diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..f88e688 --- /dev/null +++ b/.env.example @@ -0,0 +1,9 @@ +# Copy to .env (ignored by .gitignore) and fill in real values for local runs and e2e. +# e2e picks a Provider by available key, in order: Claude (ANTHROPIC_API_KEY) -> DeepSeek (DEEPSEEK_API_KEY). +ANTHROPIC_API_KEY= +# Optional: custom Claude gateway URL (defaults to AgentHub if unset). +# ANTHROPIC_BASE_URL= +# DeepSeek (CI's e2e uses this key, model deepseek-v4-flash). +DEEPSEEK_API_KEY= +# Optional: custom DeepSeek gateway URL (defaults to the official https://api.deepseek.com). +# DEEPSEEK_BASE_URL= diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..bf90052 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,57 @@ +# CI: build -> style (Prettier) -> typecheck (tsc) -> unit tests (vitest) -> live e2e (DeepSeek). +# Build first: core's exports point at dist/, and cli's type resolution and runtime imports both need core's build output. +# e2e needs the repo secret DEEPSEEK_API_KEY; when absent (e.g. forks) that step self-skips and the other checks run as usual. +name: CI + +# Limit triggers to avoid duplicate runs: push runs only on main/dev; PRs always run once (no target-branch filter -- +# this repo's PRs often target integration branches rather than main/dev, and a target filter would leave them with no CI). +on: + push: + branches: [main, dev] + pull_request: + workflow_dispatch: + +concurrency: + group: ci-${{ github.workflow }}-${{ github.event_name }}-${{ github.ref }} + cancel-in-progress: true + +jobs: + ci: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v5 + + # pnpm version comes from package.json's packageManager field. + - uses: pnpm/action-setup@v4 + + - uses: actions/setup-node@v5 + with: + node-version: 24 + cache: pnpm + + - name: Install dependencies + run: pnpm install --frozen-lockfile + + - name: Build (tsup) + run: pnpm build + + - name: Code style (Prettier) + run: pnpm format:check + + - name: Typecheck (tsc) + run: pnpm typecheck + + - name: Unit tests (vitest) + run: pnpm test + + # The secret is exposed only in this step (step-level env); earlier steps and third-party actions can't see it. + # The secrets context can't be used in if expressions, so skip inside the shell when it's absent (e.g. forks). + - name: E2E (live LLM via DeepSeek) + env: + DEEPSEEK_API_KEY: ${{ secrets.DEEPSEEK_API_KEY }} + run: | + if [ -z "$DEEPSEEK_API_KEY" ]; then + echo "DEEPSEEK_API_KEY not available; skipping e2e." + exit 0 + fi + pnpm test:e2e diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml new file mode 100644 index 0000000..357679b --- /dev/null +++ b/.github/workflows/pages.yml @@ -0,0 +1,75 @@ +# Public site: build the landing page (site root) and the docs site (/docs/) with Vite +# and deploy them as ONE artifact to GitHub Pages (scripts/build-site.mjs assembles the +# tree — landing dist with the docs dist copied under docs/). +# - Pull requests touching either package only BUILD (validation) — no Pages +# configuration and no deploy, so PRs never fail on Pages availability. +# - Pushes to main (and manual dispatch) build + deploy. BASE_PATH is derived from the +# repository name so project-pages URLs (https://.github.io//) resolve; +# repos served from a custom domain / user pages can set it to "/" instead. +# First-time setup: repository Settings -> Pages -> Source = "GitHub Actions". +name: Deploy Site + +on: + push: + branches: [main] + paths: + - "packages/landing/**" + - "packages/docs/**" + - "scripts/build-site.mjs" + - ".github/workflows/pages.yml" + pull_request: + paths: + - "packages/landing/**" + - "packages/docs/**" + - "scripts/build-site.mjs" + - ".github/workflows/pages.yml" + workflow_dispatch: + +permissions: + contents: read + pages: write + id-token: write + +# One deploy at a time; let an in-flight production deploy finish rather than cancelling it. +concurrency: + group: pages-${{ github.event_name }}-${{ github.ref }} + cancel-in-progress: false + +jobs: + build: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v5 + + # pnpm version comes from package.json's packageManager field. + - uses: pnpm/action-setup@v4 + + - uses: actions/setup-node@v5 + with: + node-version: 24 + cache: pnpm + + - name: Install dependencies + run: pnpm install --frozen-lockfile + + - name: Build site (landing + docs, vite) + env: + BASE_PATH: /${{ github.event.repository.name }}/ + run: pnpm build:site + + # Only the deploying runs need the Pages artifact (PRs stop at the build check). + - if: github.event_name != 'pull_request' + uses: actions/upload-pages-artifact@v3 + with: + path: packages/landing/dist + + deploy: + if: github.event_name != 'pull_request' + needs: build + runs-on: ubuntu-latest + environment: + name: github-pages + url: ${{ steps.deployment.outputs.page_url }} + steps: + - id: deployment + uses: actions/deploy-pages@v4 diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml new file mode 100644 index 0000000..175f648 --- /dev/null +++ b/.github/workflows/release.yml @@ -0,0 +1,236 @@ +# Release: tag v* -> build the one-line install artifacts and publish a GitHub Release. +# Two parallel jobs: +# - release: build the monorepo -> pnpm deploy a production CLI dir -> assemble penguin/ (bin + lib + web) +# -> four platform packages each bundling the official Node runtime + a universal package -> SHA256 files -> upload to the Release. +# Artifacts: penguin-{linux,darwin}-{x64,arm64}.tar.gz, penguin-universal.tar.gz, +# their .sha256 files, SHA256SUMS, and install.sh; one version per tag, multiple versions coexist. +# - publish-npm: publish the whole chain (@prismshadow/penguin-skills -> @prismshadow/penguin-core +# -> @prismshadow/penguin-server -> @prismshadow/penguin-cli) to npm at the tag version. +# skills/core serve the penguin-sdk Skill's `npm install`; server ships the built web assets inside +# the package (web-dist/, its default web dir falls back to it), so `npm install -g +# @prismshadow/penguin-cli` alone yields a working `penguin` incl. the Web UI (needs Node >= 24). +# The publish flow mirrors AgentHub's publish.yml: OIDC trusted publishing (environment: npm + +# id-token: write, no token). A Trusted Publisher can only be configured in the settings page of a +# package that ALREADY EXISTS on the registry, so a brand-new package cannot be first-published by +# this workflow. Release checklist for a new package: (1) a maintainer bootstrap-publishes it once +# manually with a one-off granular token (revoke it afterwards), running the same prepare steps as +# this job (stamp versions, build, copy LICENSE + web-dist) and publishing with `pnpm publish +# --access public --no-git-checks` -- NEVER `npm publish`, which keeps workspace:* deps unrewritten +# and yields a package that fails to install (Unsupported URL Type "workspace:"); +# (2) configure this repo + workflow as its Trusted Publisher on npmjs; (3) subsequent tags publish +# via OIDC. The publish step is idempotent (versions already on the registry are skipped), so a tag +# that failed mid-chain can be re-run as-is after fixing the config. +# npm publishing lives here rather than a separate `on: release: published` workflow: the Release is created +# by this workflow's GITHUB_TOKEN, and GitHub won't trigger other workflows' release events from that. +name: Release + +on: + push: + tags: ["v*"] + workflow_dispatch: + inputs: + tag: + description: "Release tag (e.g. v0.1.0)" + required: true + +env: + # Bundled Node runtime version (official nodejs.org dist, aligned with engines >=24). + NODE_RUNTIME_VERSION: v24.18.0 + +jobs: + release: + runs-on: ubuntu-latest + permissions: + contents: write + steps: + # On manual dispatch, check out the tag itself (not the selected branch HEAD): when re-uploading an existing + # tag's artifacts this keeps them in sync with the tag's source. On tag-push, leaving ref empty is the default. + - uses: actions/checkout@v5 + with: + ref: ${{ github.event_name == 'workflow_dispatch' && inputs.tag || '' }} + + # pnpm version comes from package.json's packageManager field. + - uses: pnpm/action-setup@v4 + + - uses: actions/setup-node@v5 + with: + node-version: 24 + cache: pnpm + + - name: Install dependencies + run: pnpm install --frozen-lockfile + + # Inject the release tag into core's VERSION constant (the source for CLI --version and the install-complete + # message); otherwise artifacts always carry the in-repo dev version and multiple installs can't be told apart. + - name: Stamp release version + run: | + TAG="${{ github.event_name == 'workflow_dispatch' && inputs.tag || github.ref_name }}" + V="${TAG#v}" + grep -q 'export const VERSION = "' packages/core/src/index.ts + sed -i "s/export const VERSION = \"[^\"]*\"/export const VERSION = \"$V\"/" packages/core/src/index.ts + + - name: Build (tsup + vite) + run: pnpm build + + # lib/: the CLI and its production deps (including workspace core/server/skills, all build outputs); + # pnpm 10's deploy needs --legacy (this repo doesn't enable inject-workspace-packages). + # bin/penguin launcher: resolve its own real path (following symlinks) -> default PENGUIN_WEB_DIST to + # the sibling web/ -> use the bundled runtime (node/bin/node) if present, else fall back to system node. + - name: Assemble penguin/ (lib + web + bin) + run: | + pnpm --filter @prismshadow/penguin-cli --prod deploy --legacy "$PWD/out/penguin/lib" + cp -r packages/web/dist out/penguin/web + mkdir -p out/penguin/bin + cat > out/penguin/bin/penguin <<'EOF' + #!/bin/sh + SELF="$0" + while [ -h "$SELF" ]; do + DIR="$(cd "$(dirname "$SELF")" && pwd)" + SELF="$(readlink "$SELF")" + case "$SELF" in /*) ;; *) SELF="$DIR/$SELF" ;; esac + done + DIR="$(cd "$(dirname "$SELF")/.." && pwd)" + export PENGUIN_WEB_DIST="${PENGUIN_WEB_DIST:-$DIR/web}" + if [ -x "$DIR/node/bin/node" ]; then + exec "$DIR/node/bin/node" "$DIR/lib/dist/index.js" "$@" + fi + exec node "$DIR/lib/dist/index.js" "$@" + EOF + chmod +x out/penguin/bin/penguin + + # Platform packages: linux uses .tar.xz, darwin uses .tar.gz (nodejs.org naming); + # node/ is only lightly trimmed (drop share/doc and share/man, keep the rest). + - name: Package platform + universal tarballs + run: | + mkdir -p dist-artifacts + for target in linux-x64 linux-arm64 darwin-x64 darwin-arm64; do + os="${target%%-*}" + arch="${target#*-}" + name="node-$NODE_RUNTIME_VERSION-$os-$arch" + if [ "$os" = "linux" ]; then ext="tar.xz"; else ext="tar.gz"; fi + curl -fsSL "https://nodejs.org/dist/$NODE_RUNTIME_VERSION/$name.$ext" -o "/tmp/$name.$ext" + rm -rf /tmp/node-runtime out/penguin/node + mkdir -p /tmp/node-runtime + if [ "$ext" = "tar.xz" ]; then + tar -xJf "/tmp/$name.$ext" -C /tmp/node-runtime + else + tar -xzf "/tmp/$name.$ext" -C /tmp/node-runtime + fi + mv "/tmp/node-runtime/$name" out/penguin/node + rm -rf out/penguin/node/share/doc out/penguin/node/share/man + tar -czf "dist-artifacts/penguin-$os-$arch.tar.gz" -C out penguin + done + # Universal package: no bundled runtime, requires system Node >= 24. + rm -rf out/penguin/node + tar -czf dist-artifacts/penguin-universal.tar.gz -C out penguin + + # SHA256SUMS summary + a same-named .sha256 per artifact (install.sh verifies against the latter). + - name: Generate SHA256 checksums + run: | + cd dist-artifacts + sha256sum *.tar.gz > SHA256SUMS + for f in *.tar.gz; do + sha256sum "$f" > "$f.sha256" + done + + # On tag-push use the ref name; on manual dispatch use the input tag. + - name: Publish GitHub Release + uses: softprops/action-gh-release@v2 + with: + tag_name: ${{ github.event_name == 'workflow_dispatch' && inputs.tag || github.ref_name }} + files: | + dist-artifacts/*.tar.gz + dist-artifacts/*.sha256 + dist-artifacts/SHA256SUMS + install.sh + + publish-npm: + name: Publish npm packages + runs-on: ubuntu-latest + # OIDC trusted publishing (same as AgentHub's publish.yml): no token; npmjs establishes trust via + # environment `npm` + this workflow. A publish failure doesn't affect the release job (independent, parallel). + permissions: + id-token: write + contents: read + environment: + name: npm + url: https://www.npmjs.com/package/@prismshadow/penguin-cli + steps: + - uses: actions/checkout@v5 + with: + ref: ${{ github.event_name == 'workflow_dispatch' && inputs.tag || '' }} + + - uses: pnpm/action-setup@v4 + + - uses: actions/setup-node@v5 + with: + node-version: 24 + cache: pnpm + registry-url: "https://registry.npmjs.org" + + # npm Trusted Publishing (OIDC) requires npm CLI >= 11.5.1: Node 24 ships npm 11.x, which satisfies it; + # this just asserts the version to guard against regressions -- pnpm publish's registry auth ultimately + # delegates to system npm, ordinary CI doesn't exercise OIDC (so a green run won't catch it), and too old + # a version only surfaces as an auth failure when actually publishing a tag. + - name: Assert npm supports trusted publishing (>= 11.5.1) + run: | + V="$(npm --version)" + echo "npm $V" + node -e 'const [M, m, p] = process.argv[1].split(".").map(Number); if (M < 11 || (M === 11 && (m < 5 || (m === 5 && p < 1)))) { console.error("npm " + process.argv[1] + " < 11.5.1"); process.exit(1); }' "$V" + + - name: Install dependencies + run: pnpm install --frozen-lockfile + + # Version always comes from the tag: package version and core's VERSION constant are injected together (the + # repo keeps the dev version). All published packages must bump in lockstep -- pnpm publish rewrites every + # workspace:* dep to the dependency's current version, so the versions must match. + - name: Stamp release version + run: | + TAG="${{ github.event_name == 'workflow_dispatch' && inputs.tag || github.ref_name }}" + V="${TAG#v}" + grep -q 'export const VERSION = "' packages/core/src/index.ts + sed -i "s/export const VERSION = \"[^\"]*\"/export const VERSION = \"$V\"/" packages/core/src/index.ts + (cd packages/skills && npm version --no-git-tag-version "$V") + (cd packages/core && npm version --no-git-tag-version "$V") + (cd packages/server && npm version --no-git-tag-version "$V") + (cd packages/cli && npm version --no-git-tag-version "$V") + + # Dependency order: cli bundles core's dist (tsup noExternal), server/web need core's types; + # web is built here only to be copied into the server package below. + - name: Build and test + run: | + pnpm --filter @prismshadow/penguin-skills build + pnpm --filter @prismshadow/penguin-core build + pnpm --filter @prismshadow/penguin-server build + pnpm --filter @prismshadow/penguin-web build + pnpm --filter @prismshadow/penguin-cli build + pnpm --filter @prismshadow/penguin-skills test + pnpm --filter @prismshadow/penguin-core test + pnpm --filter @prismshadow/penguin-server test + pnpm --filter @prismshadow/penguin-cli test + + # Copy LICENSE into the package dirs (the files allowlist includes it, so artifacts ship the license) + # and the built web assets into the server package (web-dist/, in its files allowlist: an npm install + # serves the Web UI from there without PENGUIN_WEB_DIST). + # Idempotent: a version already on the registry is skipped. The check tests `npm view`'s output + # rather than its exit code -- for a missing version of an existing package, older npm exits 0 and + # newer npm exits 1, but the output is non-empty only when the version exists. Re-running the tag + # after a mid-chain failure (e.g. Trusted Publisher not configured yet) picks up where it left off. + - name: Publish to npm + run: | + TAG="${{ github.event_name == 'workflow_dispatch' && inputs.tag || github.ref_name }}" + V="${TAG#v}" + cp LICENSE packages/skills/LICENSE + cp LICENSE packages/core/LICENSE + cp LICENSE packages/server/LICENSE + cp LICENSE packages/cli/LICENSE + rm -rf packages/server/web-dist + cp -r packages/web/dist packages/server/web-dist + for pkg in skills core server cli; do + name="@prismshadow/penguin-$pkg" + if [ -n "$(npm view "$name@$V" version 2>/dev/null || true)" ]; then + echo "$name@$V already on the registry, skipping." + continue + fi + pnpm --filter "$name" publish --access public --no-git-checks + done diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..ef6e0b8 --- /dev/null +++ b/.gitignore @@ -0,0 +1,31 @@ +.DS_Store + +# dependencies +node_modules/ + +# build output +dist/ +*.tsbuildinfo +# web assets copied into the server package at npm publish time +packages/server/web-dist/ + +# secrets / local config +.env +.env.* +!.env.example + +# agenthub trace cache +cache/ + +# legacy .penguin workspace symlink (no longer created; ignored for old working copies) +.penguin + +# logs +*.log + +# claude code session state (worktrees, skills) +.claude/ + +# playwright e2e artifacts +packages/web/test-results/ +packages/web/playwright-report/ diff --git a/.prettierignore b/.prettierignore new file mode 100644 index 0000000..67c523f --- /dev/null +++ b/.prettierignore @@ -0,0 +1,12 @@ +# build output and dependencies +node_modules/ +dist/ +pnpm-lock.yaml + +# docs-only area (mostly Chinese; not machine-formatted) +specs/ +*.md + +# Playwright e2e artifacts +test-results/ +playwright-report/ diff --git a/.prettierrc.json b/.prettierrc.json new file mode 100644 index 0000000..de753c5 --- /dev/null +++ b/.prettierrc.json @@ -0,0 +1,3 @@ +{ + "printWidth": 100 +} diff --git a/README.md b/README.md new file mode 100644 index 0000000..9347464 --- /dev/null +++ b/README.md @@ -0,0 +1,130 @@ +

+ PenguinHarness logo +

+ +

PenguinHarness

+ +

Efficient Self-Improving Harness for Everyone

+ +

+ Open-source, local-first infrastructure that builds AI agents for you — + from automatic agent construction to recursive self-improvement. +

+ +

+ CI + Deploy Site + License: Apache-2.0 + Node >= 24 +

+ +

+ English | 简体中文 · + Website · + Docs · + Blog +

+ +

+ + + PenguinHarness Web App — multi-session chat with live streaming tool calls + +

+ +--- + +## Why PenguinHarness + +- **Simplest Is the Best** — a deliberately minimal toolset over clean low-level interfaces: fewer tool calls, fewer tokens, complex tasks done efficiently. +- **Harness for Building Agents** — with the PenguinHarness SDK, an Agent builds complete Agent applications for you, autonomously, from scratch. +- **Harness for Recursive Self-Improvement** — with PenguinHarness Skills, an Agent evaluates and optimizes itself: benchmark, find the lost points, ship version N+1, snapshot before every round. +- **Local-first and lightweight** — 100% open source, runs on a single CPU, your data never leaves the machine. 1000+ online and local models reachable through one gateway. +- **Everything observable** — every request, tool call and approval decision lands in an append-only Trace; any Session can be resumed from it. + +## Quickstart + +Install with one command (Linux / macOS, x64 / arm64, bundled Node runtime): + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +``` + +Or via npm (requires Node >= 24; the command it installs is `penguin`): + +```bash +npm install -g @prismshadow/penguin-cli +``` + +Then launch the Web App — or stay in the terminal: + +```bash +penguin web # start the service and open http://127.0.0.1:7364 (first login: admin / admin123) +penguin server # same service, headless + +# configure a model once (or use the in-app Models page) +penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default + +penguin run -m "Create hello.txt containing Hello, Penguin" # one-shot task +penguin chat # interactive REPL (/compact, /exit, Ctrl-C to interrupt) +``` + +Using the SDK directly: + +```ts +import { createAgent, isCompleteModelMessage, userText } from "@prismshadow/penguin-core"; + +const agent = await createAgent({ agentId: "default_agent" }); +const session = await agent.createSession({ workspaceDir: process.cwd() }); + +for await (const output of session.run([userText("Create hello.txt containing hi")], { + approve: async () => "allow", // per-tool-call approval +})) { + if (isCompleteModelMessage(output) && output.payload.type === "text") { + console.log(output.payload.text); + } +} +``` + +## What's inside + +A pnpm monorepo (TypeScript, Node >= 24). One install ships four layers that share a single data directory (`~/.penguin/data`) and a single message protocol (OmniMessage): + +| Package | Name | Role | +| --- | --- | --- | +| [`packages/core`](packages/core) | `@prismshadow/penguin-core` | SDK & engine: ReAct loop, OmniMessage protocol, LLM/Environment interface contracts, Agent State, Trace | +| [`packages/cli`](packages/cli) | `@prismshadow/penguin-cli` | The `penguin` command: REPL, one-shot runs, model & vault config, service launcher | +| [`packages/server`](packages/server) | `@prismshadow/penguin-server` | Web backend: HTTP API + SSE streaming, multi-user auth, Project authorization, usage stats | +| [`packages/web`](packages/web) | `@prismshadow/penguin-web` | Web App: multi-session chat, Agent/skill/model management, Trace observability, evaluation center | +| [`packages/skills`](packages/skills) | `@prismshadow/penguin-skills` | Built-in skill library (agent creation, benchmarking, evaluation, optimization, …) | +| [`packages/landing`](packages/landing) | — | Product landing page (this repo's website) | +| [`packages/docs`](packages/docs) | — | Documentation site (bilingual, deployed under `/docs/`) | + +Responsibilities split by source of truth: the **SDK** owns protocol and execution (message parsing, the agent loop, tools), the **Server** owns the multi-user runtime (auth, SSE streaming, scheduled tasks), and the **file layer** under `~/.penguin/data` owns everything editable and recorded (prompts, Skills, secrets, Traces). The full design-by-design map is in [Architecture → Division of responsibilities](https://prism-shadow.github.io/penguin-harness/docs/architecture). + +## Documentation + +The docs site covers both usage and design: [Introduction](https://prism-shadow.github.io/penguin-harness/docs/) · [Quickstart](https://prism-shadow.github.io/penguin-harness/docs/quickstart) · [Architecture](https://prism-shadow.github.io/penguin-harness/docs/architecture) · [The OmniMessage Protocol](https://prism-shadow.github.io/penguin-harness/docs/omni-message) · [Core Interfaces](https://prism-shadow.github.io/penguin-harness/docs/interfaces) · [The Agent Loop](https://prism-shadow.github.io/penguin-harness/docs/agent-loop) · [CLI Reference](https://prism-shadow.github.io/penguin-harness/docs/cli) · [Server API](https://prism-shadow.github.io/penguin-harness/docs/server-api) · [Configuration](https://prism-shadow.github.io/penguin-harness/docs/configuration) + +Every doc page has a "Copy Markdown" button, so you can paste it straight into a model context. + +## Development + +```bash +pnpm install +pnpm build # build first: core's exports point at dist/ +pnpm typecheck +pnpm test + +pnpm dev:server # backend at 127.0.0.1:7364 +pnpm dev:web # web app (Vite) at 127.0.0.1:7365, /api proxied +pnpm dev:docs # docs site (Vite) at 127.0.0.1:7367 + +BASE_PATH=/ pnpm build:site # assemble landing + docs exactly like the Pages deploy +``` + +Copy `.env.example` to `.env` for model credentials in development. E2E tests run against a live model (`pnpm test:e2e`, needs `DEEPSEEK_API_KEY`). + +## License + +[Apache-2.0](LICENSE) © 2026 Prism Shadow diff --git a/README.zh.md b/README.zh.md new file mode 100644 index 0000000..7305720 --- /dev/null +++ b/README.zh.md @@ -0,0 +1,129 @@ +

+ PenguinHarness logo +

+ +

PenguinHarness

+ +

Efficient Self-Improving Harness for Everyone

+ +

+ 开源、本地优先的 AI Agent 基础设施——从自动构建 Agent 到递归自我进化。 +

+ +

+ CI + Deploy Site + License: Apache-2.0 + Node >= 24 +

+ +

+ English | 简体中文 · + 官网 · + 文档 · + 博客 +

+ +

+ + + PenguinHarness Web App——多 Session 对话与实时流式工具调用 + +

+ +--- + +## 为什么选择 PenguinHarness + +- **Simplest Is the Best**——在干净的底层接口之上刻意保持极简的工具集:更少的工具调用、更少的 Token,高效完成复杂任务。 +- **Harness for Building Agents**——基于 PenguinHarness SDK,由一个 Agent 从零开始为你自主构建完整的 Agent 应用。 +- **Harness for Recursive Self-Improvement**——基于 PenguinHarness Skills,Agent 评估并优化自己:跑 Benchmark、找失分点、产出 N+1 版本,每轮之前先做快照。 +- **本地优先且轻量**——100% 开源,一颗 CPU 即可运行,数据不出机器;经统一网关可接入 1000+ 在线与本地模型。 +- **全量可观测**——每次请求、工具调用与审批决策都以追加方式写入 Trace,任何 Session 均可从 Trace 恢复。 + +## 快速开始 + +一行命令安装(Linux / macOS,x64 / arm64,内嵌 Node 运行时,解压即用): + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +``` + +或经 npm 安装(需系统 Node >= 24,安装后的命令为 `penguin`): + +```bash +npm install -g @prismshadow/penguin-cli +``` + +然后启动 Web App,或直接留在终端: + +```bash +penguin web # 启动服务并打开 http://127.0.0.1:7364(初始账号 admin / admin123) +penguin server # 同一服务,无头运行 + +# 先配置一次模型(也可在 Web 的模型页完成) +penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default + +penguin run -m "创建 hello.txt,内容为 Hello, Penguin" # 单次任务 +penguin chat # 交互式 REPL(/compact、/exit,Ctrl-C 中断) +``` + +直接使用 SDK: + +```ts +import { createAgent, isCompleteModelMessage, userText } from "@prismshadow/penguin-core"; + +const agent = await createAgent({ agentId: "default_agent" }); +const session = await agent.createSession({ workspaceDir: process.cwd() }); + +for await (const output of session.run([userText("创建 hello.txt 并写入 hi")], { + approve: async () => "allow", // 逐个工具审批 +})) { + if (isCompleteModelMessage(output) && output.payload.type === "text") { + console.log(output.payload.text); + } +} +``` + +## 仓库结构 + +pnpm monorepo(TypeScript,Node >= 24)。一次安装交付四层组件,共享同一数据目录(`~/.penguin/data`)与同一消息协议(OmniMessage): + +| 目录 | 包名 | 职责 | +| --- | --- | --- | +| [`packages/core`](packages/core) | `@prismshadow/penguin-core` | SDK 与引擎:ReAct 循环、OmniMessage 协议、LLM/Environment 接口契约、Agent State、Trace | +| [`packages/cli`](packages/cli) | `@prismshadow/penguin-cli` | `penguin` 命令:REPL、单次运行、模型与 Vault 配置、服务启动 | +| [`packages/server`](packages/server) | `@prismshadow/penguin-server` | Web 服务端:HTTP API + SSE 流式、多用户认证、Project 授权、用量统计 | +| [`packages/web`](packages/web) | `@prismshadow/penguin-web` | Web App:多 Session 对话、Agent/技能/模型管理、Trace 观测、评估中心 | +| [`packages/skills`](packages/skills) | `@prismshadow/penguin-skills` | 内置技能库(Agent 创建、Benchmark 设计、评估、优化等) | +| [`packages/landing`](packages/landing) | — | 产品落地页(本仓库官网) | +| [`packages/docs`](packages/docs) | — | 文档站(双语,部署于 `/docs/` 路径) | + +职责按事实来源划分:**SDK** 负责协议与执行(消息解析、运行循环、工具),**Server** 负责多用户运行时(认证、SSE 流式、定时任务),`~/.penguin/data` 下的**文件层**承载一切可编辑与被记录的状态(Prompt、Skill、密钥、Trace)。逐项对应表见[架构总览 → 职责划分](https://prism-shadow.github.io/penguin-harness/docs/architecture)。 + +## 文档 + +文档站覆盖使用与设计两个层面:[产品介绍](https://prism-shadow.github.io/penguin-harness/docs/) · [快速开始](https://prism-shadow.github.io/penguin-harness/docs/quickstart) · [架构总览](https://prism-shadow.github.io/penguin-harness/docs/architecture) · [OmniMessage 协议](https://prism-shadow.github.io/penguin-harness/docs/omni-message) · [接口契约](https://prism-shadow.github.io/penguin-harness/docs/interfaces) · [Agent 运行循环](https://prism-shadow.github.io/penguin-harness/docs/agent-loop) · [CLI 参考](https://prism-shadow.github.io/penguin-harness/docs/cli) · [Server API](https://prism-shadow.github.io/penguin-harness/docs/server-api) · [配置参考](https://prism-shadow.github.io/penguin-harness/docs/configuration) + +每页文档都带「复制 Markdown」按钮,可直接粘贴进模型上下文。 + +## 本地开发 + +```bash +pnpm install +pnpm build # 先构建:core 的导出指向 dist/ +pnpm typecheck +pnpm test + +pnpm dev:server # 服务端 127.0.0.1:7364 +pnpm dev:web # Web App(Vite)127.0.0.1:7365,/api 代理到服务端 +pnpm dev:docs # 文档站(Vite)127.0.0.1:7367 + +BASE_PATH=/ pnpm build:site # 按 Pages 部署的方式组装 落地页 + 文档 +``` + +开发态模型凭据可复制 `.env.example` 为 `.env` 填写。E2E 测试走真实模型(`pnpm test:e2e`,需要 `DEEPSEEK_API_KEY`)。 + +## 许可证 + +[Apache-2.0](LICENSE) © 2026 Prism Shadow diff --git a/install.sh b/install.sh new file mode 100644 index 0000000..db42e16 --- /dev/null +++ b/install.sh @@ -0,0 +1,153 @@ +#!/bin/sh +# PenguinHarness one-line installer. +# +# curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +# +# Options: +# PENGUIN_VERSION=vX.Y.Z pin a version (same as --version vX.Y.Z); default is the latest Release +# PENGUIN_INSTALL_DIR= install dir; default ~/.penguin +# --universal install the universal package (no bundled Node runtime; needs system Node >= 24) +# +# The data dir (~/.penguin/data) sits under the install home but is never touched by reinstall/upgrade (which only replace bin/lib/web/node). +# +# Docs: https://prism-shadow.github.io/penguin-harness/docs/installation +set -eu + +REPO="https://github.com/Prism-Shadow/penguin-harness" +VERSION="${PENGUIN_VERSION:-}" +INSTALL_DIR="${PENGUIN_INSTALL_DIR:-$HOME/.penguin}" +BIN_DIR="$HOME/.local/bin" +UNIVERSAL=0 + +fail() { + echo "error: $1" >&2 + exit 1 +} + +# --- Parse args (also passable via curl | sh -s -- --universal) --- +while [ $# -gt 0 ]; do + case "$1" in + --version) + [ $# -ge 2 ] || fail "--version requires a value (e.g. --version v1.0.0)" + VERSION="$2" + shift 2 + ;; + --universal) + UNIVERSAL=1 + shift + ;; + *) + fail "unknown option: $1" + ;; + esac +done + +# --- Detect platform: Linux/Darwin x64/arm64; other platforms should use the universal package --- +ASSET="penguin-universal.tar.gz" +if [ "$UNIVERSAL" -eq 0 ]; then + case "$(uname -s)" in + Linux) os="linux" ;; + Darwin) os="darwin" ;; + *) fail "unsupported OS: $(uname -s). Install Node.js >= 24, then re-run with --universal." ;; + esac + case "$(uname -m)" in + x86_64) arch="x64" ;; + aarch64 | arm64) arch="arm64" ;; + *) fail "unsupported architecture: $(uname -m). Install Node.js >= 24, then re-run with --universal." ;; + esac + ASSET="penguin-$os-$arch.tar.gz" +fi + +# --- Universal package precheck: system Node >= 24 (platform packages bundle the runtime, so exempt) --- +if [ "$UNIVERSAL" -eq 1 ]; then + command -v node >/dev/null 2>&1 \ + || fail "the universal package needs Node.js >= 24 on PATH (none found)." + node_version="$(node --version)" # e.g. v24.18.0 + v="${node_version#v}" + major="${v%%.*}" + if [ "$major" -lt 24 ]; then + fail "the universal package needs Node.js >= 24, found $node_version." + fi +fi + +# --- Download (latest Release by default; PENGUIN_VERSION pins a version) --- +if [ -n "$VERSION" ]; then + BASE_URL="$REPO/releases/download/$VERSION" +else + BASE_URL="$REPO/releases/latest/download" +fi +TMP="$(mktemp -d)" +trap 'rm -rf "$TMP"' EXIT + +echo "Downloading $BASE_URL/$ASSET ..." +curl -fSL --progress-bar "$BASE_URL/$ASSET" -o "$TMP/$ASSET" \ + || fail "download failed. Check the version tag and your network, then retry." + +# --- SHA256 verify: only when .sha256 exists (skip on 404); warn and skip if no checksum tool --- +if curl -fsSL "$BASE_URL/$ASSET.sha256" -o "$TMP/$ASSET.sha256" 2>/dev/null; then + if command -v sha256sum >/dev/null 2>&1; then + (cd "$TMP" && sha256sum -c "$ASSET.sha256" >/dev/null 2>&1) || fail "checksum mismatch for $ASSET." + echo "Checksum OK." + elif command -v shasum >/dev/null 2>&1; then + (cd "$TMP" && shasum -a 256 -c "$ASSET.sha256" >/dev/null 2>&1) || fail "checksum mismatch for $ASSET." + echo "Checksum OK." + else + echo "warning: sha256sum/shasum not found; skipping checksum verification." >&2 + fi +else + echo "warning: checksum file not available; skipping verification." >&2 +fi + +# --- Extract and swap into place: first move the new dirs into a staging area inside the install +# dir (same filesystem as the final location, so any slow cross-device copy happens before the +# old install is touched), then swap fast (rm old + same-disk mv is a rename; tiny window). +# No stale files after upgrade; the universal package has no node/, so cleanup lets the wrapper +# fall back to system Node. The data dir (~/.penguin/data) is untouched. --- +tar -xzf "$TMP/$ASSET" -C "$TMP" +[ -d "$TMP/penguin" ] || fail "unexpected archive layout: top-level penguin/ missing." +mkdir -p "$INSTALL_DIR" +STAGING="$INSTALL_DIR/.staging.$$" +rm -rf "$STAGING" +mkdir -p "$STAGING" +trap 'rm -rf "$TMP" "$STAGING"' EXIT +for d in bin lib web node; do + if [ -e "$TMP/penguin/$d" ]; then + mv "$TMP/penguin/$d" "$STAGING/$d" + fi +done +[ -x "$STAGING/bin/penguin" ] || fail "unexpected archive layout: bin/penguin missing." +rm -rf "$INSTALL_DIR/bin" "$INSTALL_DIR/lib" "$INSTALL_DIR/web" "$INSTALL_DIR/node" +for d in bin lib web node; do + if [ -e "$STAGING/$d" ]; then + mv "$STAGING/$d" "$INSTALL_DIR/$d" + fi +done +rm -rf "$STAGING" +[ -x "$INSTALL_DIR/bin/penguin" ] || fail "install incomplete: $INSTALL_DIR/bin/penguin missing." + +# --- Symlink into ~/.local/bin and check PATH --- +mkdir -p "$BIN_DIR" +ln -sf "$INSTALL_DIR/bin/penguin" "$BIN_DIR/penguin" +case ":$PATH:" in + *":$BIN_DIR:"*) ;; + *) + echo "" + echo "note: $BIN_DIR is not on your PATH. Add it to your shell profile:" + case "${SHELL:-}" in + */zsh) echo " echo 'export PATH=\"\$HOME/.local/bin:\$PATH\"' >> ~/.zshrc && source ~/.zshrc" ;; + */bash) echo " echo 'export PATH=\"\$HOME/.local/bin:\$PATH\"' >> ~/.bashrc && source ~/.bashrc" ;; + */fish) echo " fish_add_path \$HOME/.local/bin" ;; + *) echo " export PATH=\"\$HOME/.local/bin:\$PATH\"" ;; + esac + ;; +esac + +# --- Finish: print version and getting-started tips --- +installed_version="$("$INSTALL_DIR/bin/penguin" --version 2>/dev/null || echo "unknown")" +echo "" +echo "PenguinHarness $installed_version installed to $INSTALL_DIR" +echo "" +echo "Get started:" +echo " penguin --help # all commands" +echo " penguin web # start the Web UI at http://127.0.0.1:7364 (initial login: admin / admin123)" +echo " penguin server # headless server (PORT / HOST to override)" diff --git a/package.json b/package.json new file mode 100644 index 0000000..8e98f69 --- /dev/null +++ b/package.json @@ -0,0 +1,34 @@ +{ + "name": "penguin-harness", + "version": "0.0.1", + "private": true, + "type": "module", + "description": "PenguinHarness — TypeScript AI Agent (SDK + CLI).", + "engines": { + "node": ">=24" + }, + "scripts": { + "format": "prettier --write .", + "format:check": "prettier --check .", + "typecheck": "pnpm -r typecheck", + "test": "pnpm -r test", + "test:e2e": "pnpm --filter @prismshadow/penguin-core test:e2e", + "build": "pnpm -r build && pnpm link:cli", + "link:cli": "pnpm --dir packages/cli link --global || echo '[link:cli] 未配置 pnpm 全局目录,跳过 penguin 全局链接(先运行一次 pnpm setup)'", + "penguin": "tsx packages/cli/src/index.ts", + "dev:server": "pnpm --filter @prismshadow/penguin-skills --filter @prismshadow/penguin-core build && pnpm --filter @prismshadow/penguin-server dev", + "dev:web": "pnpm --filter @prismshadow/penguin-skills --filter @prismshadow/penguin-core build && pnpm --filter @prismshadow/penguin-web dev", + "dev:docs": "pnpm --filter @prismshadow/penguin-docs dev", + "build:site": "node scripts/build-site.mjs" + }, + "devDependencies": { + "@prismshadow/penguin-cli": "workspace:*", + "@prismshadow/penguin-core": "workspace:*", + "@types/node": "^24.0.0", + "prettier": "^3.9.4", + "tsx": "^4.20.0", + "typescript": "^5.6.0", + "vitest": "^2.1.0" + }, + "packageManager": "pnpm@10.26.2" +} diff --git a/packages/cli/README.md b/packages/cli/README.md new file mode 100644 index 0000000..de61d8b --- /dev/null +++ b/packages/cli/README.md @@ -0,0 +1,39 @@ +# @prismshadow/penguin-cli + +The PenguinHarness command line. Installs the `penguin` command: an interactive REPL, a one-shot task runner, model / vault configuration, and the launcher for the Web service. + +```bash +npm install -g @prismshadow/penguin-cli # requires Node >= 24 +``` + +```bash +penguin web # start the Web service and open http://127.0.0.1:7364 +penguin server # same service, headless + +penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default + +penguin run -m "Create hello.txt containing Hello, Penguin" # one Task, then exit +penguin chat # REPL: /compact, /exit, Ctrl-C interrupts +penguin chat --resume # resume the latest session +``` + +Tool calls go through an approval gate — `--approve allow-all` (default) `| deny-all | read-only | always-ask`. Data lives under `~/.penguin/data` (`PENGUIN_HOME` or `--root` override); model credentials come from the Project config or provider env vars (e.g. `DEEPSEEK_API_KEY`). + +Prefer a one-line install with a bundled Node runtime? See the [installation guide](https://prism-shadow.github.io/penguin-harness/docs/installation). + +## Documentation + +- [Quickstart](https://prism-shadow.github.io/penguin-harness/docs/quickstart) +- [CLI Reference](https://prism-shadow.github.io/penguin-harness/docs/cli) +- [Configuration Reference](https://prism-shadow.github.io/penguin-harness/docs/configuration) + +## Development + +```bash +pnpm penguin # run from source (repo root, via tsx) +pnpm --filter @prismshadow/penguin-cli build # tsup → dist/index.js (the penguin bin) +pnpm --filter @prismshadow/penguin-cli typecheck +pnpm --filter @prismshadow/penguin-cli test +``` + +Part of [PenguinHarness](https://github.com/Prism-Shadow/penguin-harness) · Apache-2.0 diff --git a/packages/cli/package.json b/packages/cli/package.json new file mode 100644 index 0000000..9db412e --- /dev/null +++ b/packages/cli/package.json @@ -0,0 +1,48 @@ +{ + "name": "@prismshadow/penguin-cli", + "version": "0.0.1", + "type": "module", + "description": "PenguinHarness CLI: interactive REPL and single-task runner over @prismshadow/penguin-core.", + "license": "Apache-2.0", + "repository": { + "type": "git", + "url": "git+https://github.com/Prism-Shadow/penguin-harness.git", + "directory": "packages/cli" + }, + "bin": { + "penguin": "./dist/index.js" + }, + "engines": { + "node": ">=24" + }, + "scripts": { + "typecheck": "tsc --noEmit -p tsconfig.json", + "test": "vitest run --passWithNoTests", + "build": "tsup", + "penguin": "tsx src/index.ts" + }, + "dependencies": { + "@prismshadow/agenthub": "^0.3.3", + "@prismshadow/penguin-core": "workspace:*", + "@prismshadow/penguin-server": "workspace:*", + "@prismshadow/penguin-skills": "workspace:*", + "commander": "^13.0.0", + "dotenv": "^17.0.0", + "smol-toml": "^1.3.0", + "yaml": "^2.5.0" + }, + "devDependencies": { + "@types/node": "^24.0.0", + "tsup": "^8.3.0", + "tsx": "^4.20.0", + "typescript": "^5.6.0", + "vitest": "^2.1.0" + }, + "files": [ + "dist", + "LICENSE" + ], + "publishConfig": { + "access": "public" + } +} diff --git a/packages/cli/src/approval.ts b/packages/cli/src/approval.ts new file mode 100644 index 0000000..54761ec --- /dev/null +++ b/packages/cli/src/approval.ts @@ -0,0 +1,142 @@ +/** + * CLI tool-call approval. + * + * The CLI consumes the output stream of `session.run()`; within a turn, the engine invokes the + * injected `approve` callback for each tool_call. + * Docs: /docs/cli § "Approval modes (--approve)"; /docs/tools § "Approval". + */ +import { createInterface } from "node:readline"; +import type { ApprovalDecision, ApproveFn } from "@prismshadow/penguin-core"; +import { defaultMessages } from "./i18n.js"; +import type { Messages } from "./i18n.js"; + +/** Valid string values for the `--approve` option (includes the default allow-all, so scripts can specify it explicitly and get the default behavior). */ +const APPROVE_MODES = ["allow-all", "deny-all", "read-only", "always-ask"] as const; + +/** + * Approval mode (derived from APPROVE_MODES, the single source of truth): + * - `allow-all`: auto-approve every tool (default mode); + * - `deny-all`: auto-reject every tool; + * - `read-only`: auto-approve read-only tools (permission === "r"), defer the rest to a human; + * - `always-ask`: interactive approval for each call. + */ +export type ApprovalMode = (typeof APPROVE_MODES)[number]; + +/** + * Resolve the approval mode from the CLI: read `--approve`, default `allow-all`; print a + * message and exit if the value is invalid. + */ +export function resolveApprovalMode(approve: string | undefined, t: Messages): ApprovalMode { + if (approve === undefined) return "allow-all"; + const v = approve.trim().toLowerCase(); + if ((APPROVE_MODES as readonly string[]).includes(v)) { + return v as ApprovalMode; + } + process.stderr.write(`${t.approveModeInvalid(approve)}\n`); + process.exit(1); +} + +/** + * Build the `approve` callback for a given permission mode. `toolPermission` looks up a tool's + * permission level; `interactivePrompt` is the actual Q&A used when deferring to a human (run + * uses a one-off prompt, chat uses a persistent readline). Rendering the approval result is not + * done here — `context_engine` emits the decision as an `approval_decision` event, rendered by + * the frontend (see render.ts). + */ +export function makeApprove(args: { + mode: ApprovalMode; + toolPermission: (name: string) => "r" | "rw" | undefined; + interactivePrompt: ApproveFn; +}): ApproveFn { + const { mode, toolPermission, interactivePrompt } = args; + return async (toolCall) => { + const name = toolCall.payload.name; + switch (mode) { + case "allow-all": + return "allow"; + case "deny-all": + return "deny"; + case "read-only": + // Auto-approve read-only tools; defer read-write/unknown tools to a human. + if (toolPermission(name) === "r") return "allow"; + return interactivePrompt(toolCall); + case "always-ask": + default: + return interactivePrompt(toolCall); + } + }; +} + +export interface PromptApprovalOptions { + /** Message set; resolved from the env var by default. */ + t?: Messages; + /** Stream to read the approval answer from; defaults to `process.stdin`. */ + input?: NodeJS.ReadableStream; + /** Stream to print the approval prompt to; defaults to `process.stdout`. */ + output?: NodeJS.WritableStream; +} + +/** + * Callback that rejects the pending approval while `promptApproval` is waiting; `null` + * otherwise. `penguin run` calls `denyActivePrompt()` from a single global SIGINT handler: + * Ctrl-C during approval collapses to "deny" (consistent with chat), and only interrupts the + * whole turn at other times. SIGINT is registered in exactly one place (run); promptApproval no + * longer attaches its own listener. + */ +let activePromptDeny: (() => void) | null = null; +export function denyActivePrompt(): boolean { + if (!activePromptDeny) return false; + activePromptDeny(); + return true; +} + +/** + * One-off interactive approval Q&A (for non-persistent REPL scenarios like `run`). The pending + * tool call has already been streamed above. Input-stream EOF/close is treated as a deny; + * Ctrl-C while waiting is collapsed to a deny by the caller (run) via `denyActivePrompt`. The + * readline instance is closed after reading. + */ +export function promptApproval(opts: PromptApprovalOptions = {}): Promise { + const input = opts.input ?? process.stdin; + const output = opts.output ?? process.stdout; + const t = opts.t ?? defaultMessages(); + const rl = createInterface({ input, output }); + return new Promise((resolve) => { + let resolved = false; + const finish = (decision: ApprovalDecision) => { + if (resolved) return; + resolved = true; + // Only clear our own slot: even under concurrent prompts (upstream already serializes + // this; this is a defensive check), don't clobber someone else's deny hook. + if (activePromptDeny === deny) activePromptDeny = null; + rl.close(); + resolve(decision); + }; + const deny = () => finish("deny"); + // Ctrl-C during approval is turned into a "deny" via run's global SIGINT calling + // denyActivePrompt (no duplicate SIGINT listener registered here); input-stream EOF/close + // is likewise treated as a deny, to avoid hanging. + activePromptDeny = deny; + rl.on("close", () => finish("deny")); + rl.question(t.approvePrompt(), (answer) => { + // Tool approval defaults to allow: a bare Enter (empty input) counts as allow. + finish(parseApprovalAnswer(answer, "allow")); + }); + }); +} + +/** + * Parse an approval/confirmation answer (trimmed, case-insensitive): `y`/`yes` → allow, + * `n`/`no` → deny, everything else (including empty input/bare Enter) → `fallback`. Tool + * approval defaults to allow (pass `"allow"`); exit/restart-style confirmations default to no + * (the default `"deny"`). + */ +export function parseApprovalAnswer( + answer: string, + fallback: ApprovalDecision = "deny", +): ApprovalDecision { + const normalized = answer.trim().toLowerCase(); + if (normalized === "y" || normalized === "yes") return "allow"; + if (normalized === "n" || normalized === "no") return "deny"; + return fallback; +} diff --git a/packages/cli/src/commands/chat.ts b/packages/cli/src/commands/chat.ts new file mode 100644 index 0000000..5873ae6 --- /dev/null +++ b/packages/cli/src/commands/chat.ts @@ -0,0 +1,369 @@ +/** + * `penguin chat` — interactive REPL. + * + * penguin chat [--model-id ] [--provider ] [--project-id ] [--agent-id ] + * [--workspace ] [--approve ] + * + * Each line of input starts one conversation turn; `/compact` proactively compacts the + * context (reason=manual); `/exit` or `/quit` exits. + * Uses the current directory when no Workspace is specified. + * + * Multi-line input: trailing `\` continues the line; when the terminal supports bracketed + * paste, a multi-line paste is treated as a single message (sent on Enter). + * + * Ctrl-C behavior (state-dependent): buffer has content -> clear it; + * awaiting approval -> deny; running -> abort the current turn and return to input; + * empty buffer -> show a y/N exit confirmation. + * + * Implementation notes: on a TTY, stdin is put into raw mode with bracketed paste enabled; + * stdin is piped through PasteFilter into a readline created with `terminal: true` — Ctrl-C + * is captured in-process by readline as 'SIGINT' (it never escapes as an OS signal killing + * the process group), and pasted content is held whole by PasteFilter (not split into + * multiple submits by embedded newlines). + * Docs: /docs/cli § "penguin chat". + */ +import { createInterface, type Interface } from "node:readline"; +import type { Command } from "commander"; +import { createAgent, userText } from "@prismshadow/penguin-core"; +import type { ApprovalDecision, OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core"; +import { StreamRenderer, dim, renderHistory } from "../render.js"; +import { runTask } from "../task-loop.js"; +import { parseApprovalAnswer, resolveApprovalMode } from "../approval.js"; +import { LineComposer, PasteFilter } from "../input.js"; +import type { Messages } from "../i18n.js"; + +export type ChatState = "idle" | "running" | "approving" | "confirming-exit"; + +export type SigintAction = "deny" | "abort" | "clear" | "confirm-exit" | "exit"; + +/** Pure decision: current state + whether the input buffer is non-empty -> the action Ctrl-C should perform. */ +export function decideSigint(state: ChatState, hasBufferedInput: boolean): SigintAction { + if (state === "approving") return "deny"; + if (state === "running") return "abort"; + if (state === "confirming-exit") return "exit"; + return hasBufferedInput ? "clear" : "confirm-exit"; +} + +interface RlInternals { + line: string; + cursor: number; + _refreshLine?: () => void; +} + +const MAIN_PROMPT = "> "; +const CONT_PROMPT = "… "; + +export function registerChatCommand(program: Command, t: Messages): void { + program + .command("chat") + .description(t.chat.desc) + .option("--model-id ", t.common.modelId) + .option("--provider ", t.common.provider) + .option("--project-id ", t.common.projectId) + .option("--agent-id ", t.common.agentId) + .option("--workspace ", t.common.workspace) + .option("--approve ", t.common.approve) + .option("--resume [sessionId]", t.chat.resume) + .action(async (opts) => { + const mode = resolveApprovalMode(opts.approve, t); + const out = process.stdout; + + const agent = await createAgent({ + ...(opts.agentId ? { agentId: opts.agentId } : {}), + ...(opts.projectId ? { projectId: opts.projectId } : {}), + }); + + // --resume: resumes an existing Session. Workspace and + // Model follow the original Session and cannot be overridden; when omitted, resumes + // the current Agent's most recent Session. + let session; + if (opts.resume !== undefined) { + if (opts.workspace || opts.modelId || opts.provider) { + out.write(`${t.error(t.resumeNoOverride())}\n`); + process.exitCode = 1; + return; + } + const sessionId = + typeof opts.resume === "string" ? opts.resume : await agent.latestSessionId(); + if (!sessionId) { + out.write(`${t.error(t.resumeNoSession())}\n`); + process.exitCode = 1; + return; + } + session = await agent.resumeSession({ sessionId }); + } else { + session = await agent.createSession({ + workspaceDir: opts.workspace ?? process.cwd(), + ...(opts.modelId ? { modelId: opts.modelId } : {}), + ...(opts.provider ? { provider: opts.provider } : {}), + }); + } + + const renderer = new StreamRenderer(out, t); + + out.write( + `${t.header("chat", agent.state.agentId, session.workspaceDir, session.modelId)}\n` + + `${t.chatHints()}\n`, + ); + // On resume, first render the history messages of the current context per Trace + // (full messages, including interrupted turns and their markers), then proceed to + // regular input. + if (session.resumedHistory) { + out.write(`${t.resumedBanner(session.sessionId, session.resumedHistory.length)}\n`); + renderHistory(session.resumedHistory, out); + } + + // TTY: raw mode + bracketed paste + PasteFilter; non-TTY (pipe/test): read stdin directly. + const isTTY = Boolean(process.stdin.isTTY); + let pasteFilter: PasteFilter | null = null; + let inputStream: NodeJS.ReadableStream = process.stdin; + if (isTTY) { + process.stdin.setRawMode(true); + out.write("\x1b[?2004h"); + pasteFilter = new PasteFilter(); + process.stdin.pipe(pasteFilter); + inputStream = pasteFilter; + } + + const rl = createInterface({ + input: inputStream, + output: out, + terminal: isTTY, + }); + const rli = rl as unknown as RlInternals; + const composer = new LineComposer(); + + let state: ChatState = "idle"; + let closed = false; + let taskAbort: AbortController | null = null; + let pendingLine: ((line: string | null) => void) | null = null; + let pendingApproval: ((decision: ApprovalDecision) => void) | null = null; + + const cleanup = () => { + if (!isTTY) return; + try { + out.write("\x1b[?2004l"); + } catch { + /* ignore */ + } + try { + process.stdin.setRawMode(false); + } catch { + /* ignore */ + } + try { + if (pasteFilter) process.stdin.unpipe(pasteFilter); + } catch { + /* ignore */ + } + try { + process.stdin.pause(); + } catch { + /* ignore */ + } + }; + process.once("exit", cleanup); + + if (pasteFilter) { + pasteFilter.on("paste", (text: string) => { + if (state !== "idle") return; // ignore paste while running + const { lineCount, normalized } = composer.pushPaste(text); + if (lineCount === 0) return; + out.write(`${normalized}\n`); + rl.setPrompt(CONT_PROMPT); + rl.prompt(); + }); + } + + rl.on("line", (line) => { + if (state === "confirming-exit") { + if (parseApprovalAnswer(line) === "allow") { + rl.close(); + } else { + state = "idle"; + composer.reset(); + out.write("\n"); + rl.setPrompt(MAIN_PROMPT); + rl.prompt(); + } + return; + } + if (state === "idle" && pendingLine) { + const { message } = composer.pushTypedLine(line); + if (message === undefined) { + // Continuation: show the continuation prompt and keep waiting. + rl.setPrompt(CONT_PROMPT); + rl.prompt(); + } else { + const resolve = pendingLine; + pendingLine = null; + resolve(message); + } + return; + } + if (state === "approving" && pendingApproval) { + const resolve = pendingApproval; + pendingApproval = null; + // Tool approval defaults to allow: pressing Enter (empty input) is treated as allow. + resolve(parseApprovalAnswer(line, "allow")); + } + // running: ignore any line typed at this moment. + }); + + rl.on("SIGINT", () => { + const hasBuffer = rli.line.length > 0 || composer.hasPending(); + const action = decideSigint(state, hasBuffer); + if (action === "deny") { + if (pendingApproval) { + const resolve = pendingApproval; + pendingApproval = null; + out.write("\n"); + resolve("deny"); + } + } else if (action === "abort") { + if (taskAbort && !taskAbort.signal.aborted) { + out.write(`\n${t.taskInterrupted()}\n`); + taskAbort.abort(); + } + } else if (action === "clear") { + composer.reset(); + rl.setPrompt(MAIN_PROMPT); + clearCurrentLine(rl, rli, out); + } else if (action === "confirm-exit") { + state = "confirming-exit"; + rli.line = ""; + rli.cursor = 0; + out.write("\n"); + rl.setPrompt(t.confirmExit()); + rl.prompt(); + } else { + out.write("\n"); + rl.close(); + } + }); + + rl.on("close", () => { + closed = true; + if (pendingLine) { + const resolve = pendingLine; + pendingLine = null; + resolve(null); + } + }); + + const askLine = (): Promise => + new Promise((resolve) => { + if (closed) { + resolve(null); + return; + } + state = "idle"; + pendingLine = resolve; + composer.reset(); + rli.line = ""; + rli.cursor = 0; + out.write("\n"); + rl.setPrompt(MAIN_PROMPT); + rl.prompt(); + }); + + // Interactive approval prompt: reuses the persistent readline, prompt text is + // localized; the tool call is already rendered above via streaming, so it is not + // re-rendered here. + const interactivePrompt = (_tc: OmniMessage): Promise => + new Promise((resolve) => { + state = "approving"; + pendingApproval = (decision) => { + state = "running"; + resolve(decision); + }; + rl.setPrompt(t.approvePrompt()); + rl.prompt(); + }); + + // Whether this Session already has a resumable Trace record: a resumed Session + // naturally has one; a new Session gets one starting from its first Task / compact + // (session_meta is written along with it). This decides whether to print the resume + // command example on exit. + let resumable = opts.resume !== undefined; + + try { + for (;;) { + const line = await askLine(); + if (line === null) break; + const text = line.trim(); + if (text === "/exit" || text === "/quit") break; + if (text.length === 0) continue; + + state = "running"; + taskAbort = new AbortController(); + try { + if (text === "/compact") { + // Proactive context compaction (Task boundary, reason=manual): the renderer + // prints compaction progress; Ctrl-C aborts the compaction via signal + // (preserving the original context). When there's nothing to compact (session + // just started / two consecutive /compact calls), the engine silently returns + // and we add one line of feedback here. Afterwards, settle the renderer's + // counters (endCompact) — compaction usage is already shown on the completion + // line and must not be counted again toward the next task's stats delta. + const startedAt = Date.now(); + let sawMessage = false; + try { + for await (const msg of session.compact({ + signal: taskAbort.signal, + })) { + sawMessage = true; + resumable = true; + renderer.handle(msg); + } + } finally { + renderer.endCompact(Date.now() - startedAt); + } + if (!sawMessage) out.write(`${t.compactNothing()}\n`); + } else { + resumable = true; + await runTask(session, [userText(text)], { + mode, + signal: taskAbort.signal, + renderer, + interactivePrompt, + t, + }); + } + } catch (err) { + out.write(`\n${t.error(err instanceof Error ? err.message : String(err))}\n`); + } finally { + taskAbort = null; + state = "idle"; + } + } + } finally { + rl.close(); + cleanup(); + session.dispose(); // tear down managed long-running command sessions to avoid leaking background processes + process.removeListener("exit", cleanup); + // On exit, print a dimmed resume command example: includes this + // session's Project / Agent options so the command can be copy-pasted directly; + // skipped when the Session has no Trace record yet (nothing to resume). + if (resumable) { + const command = + `penguin chat --resume ${session.sessionId}` + + (opts.projectId ? ` --project-id ${opts.projectId}` : "") + + (opts.agentId ? ` --agent-id ${opts.agentId}` : ""); + out.write(`${dim(t.resumeHint(command))}\n`); + } + } + }); +} + +/** Clear the current input line and redraw the prompt (Ctrl-C clears the buffer when it has content). */ +function clearCurrentLine(rl: Interface, rli: RlInternals, out: NodeJS.WritableStream): void { + rli.line = ""; + rli.cursor = 0; + if (typeof rli._refreshLine === "function") { + rli._refreshLine(); + } else { + out.write("\r\x1b[K"); + rl.prompt(true); + } +} diff --git a/packages/cli/src/commands/config.ts b/packages/cli/src/commands/config.ts new file mode 100644 index 0000000..c3642ce --- /dev/null +++ b/packages/cli/src/commands/config.ts @@ -0,0 +1,368 @@ +/** + * `penguin config` — manages a Project's model credentials, default model, model list, + * Agent-level vault environment variables, and UI language. + * + * penguin config model add --model-id [--provider ] [--api-key ] [--context-window ] [--set-default] [--root ] + * penguin config model default --model-id --provider [--root ] + * penguin config model vision --model-id --provider [--root ] + * penguin config model list [--root ] + * penguin config vault set --key --value [--agent-id ] [--root ] + * penguin config vault list [--agent-id ] [--root ] + * penguin config vault remove --key [--agent-id ] [--root ] + * penguin config lang + * + * `--model-id` always takes the **upstream id** (the request id sent to AgentHub verbatim), + * which together with `--provider` forms a `(provider, model_id)` paired reference — + * **no string concatenation is ever performed**. For `model add`, --provider defaults to + * an inference from the built-in catalog (falling back to custom when inference fails); + * a new entry's client_type defaults according to the group's semantics (not set for + * first-party vendors; openai for custom / self-hosted groups / gateways, with the + * gateway's endpoint base URL pre-filled). For `model default` / `model vision`, + * --provider is **required**; core validation raises an error when the reference is not + * found in models. `--root` specifies the data root directory (priority: option > + * PENGUIN_HOME > ~/.penguin/data). The UI language is controlled by the PENGUIN_LANG + * environment variable; `config lang` writes it into the shell startup file and restarts + * the shell to take effect. + * Docs: /docs/cli § "penguin config". + */ +import { homedir } from "node:os"; +import path from "node:path"; +import { createInterface } from "node:readline"; +import type { Command } from "commander"; +import { + DEFAULT_AGENT_ID, + DEFAULT_PROJECT_ID, + type ModelPricing, + type ModelRef, + type ProjectConfig, + addModel, + catalogEntryFor, + formatModelRef, + getModel, + inferProviderForUpstream, + loadAgentVault, + loadProjectConfig, + providerInfo, + removeVaultEntry, + resolveRoot, + setDefaultModel, + setVaultEntry, + setVisionModel, +} from "@prismshadow/penguin-core"; +import { parseApprovalAnswer } from "../approval.js"; +import { getMessages, maskApiKey, type Messages } from "../i18n.js"; +import { applyLanguageToRc, restartShell } from "../lang-config.js"; + +/** Data root directory: the `--root` option takes priority (relative paths resolved against cwd), then PENGUIN_HOME / ~/.penguin/data. */ +function resolveRootOption(root: string | undefined): string { + return root !== undefined ? path.resolve(root) : resolveRoot(); +} + +/** + * Renders the model list as column-aligned lines (the default model is marked with `*`; + * fully empty columns are omitted automatically). `provider` and `model_id` each occupy + * their own column (stored fields, never split apart); `vision` reflects the effective + * semantics (the TOML `vision` annotation takes priority, falling back to the catalog + * annotation — matched by the (provider, model_id) pair — and recorded as Y under + * "default = supported" when neither is present). Exported for unit tests. + */ +export function formatModelRows(cfg: ProjectConfig): string[] { + const cells = cfg.models.map((entry) => { + const cat = catalogEntryFor(entry.provider, entry.model_id); + const vision = entry.vision ?? cat?.supportsVision ?? true; + const isDefault = + cfg.default_model?.provider === entry.provider && + cfg.default_model?.model_id === entry.model_id; + return { + provider: `${isDefault ? "* " : " "}${entry.provider}`, + model: entry.model_id, + vision: `vision=${vision ? "Y" : "-"}`, + context_window: + entry.context_window !== undefined ? `context_window=${entry.context_window}` : "", + client_type: entry.client_type ? `client_type=${entry.client_type}` : "", + pricing: entry.pricing + ? `price=${entry.pricing.cache_read}/${entry.pricing.cache_write}/${entry.pricing.output}` + : "", + api_key: `api_key=${maskApiKey(entry.api_key)}`, + base_url: entry.base_url ? `base_url=${entry.base_url}` : "", + }; + }); + const columns = [ + "provider", + "model", + "vision", + "context_window", + "client_type", + "pricing", + "api_key", + "base_url", + ] as const; + const widths = columns.map((c) => Math.max(...cells.map((cell) => cell[c].length))); + const active = columns + .map((c, i) => ({ key: c, width: widths[i]! })) + .filter((col) => col.width > 0); + return cells.map((cell) => + active + .map((col, i) => (i === active.length - 1 ? cell[col.key] : cell[col.key].padEnd(col.width))) + .join(" ") + .trimEnd(), + ); +} + +export function registerConfigCommand(program: Command, t: Messages): void { + const config = program.command("config").description(t.config.desc); + const model = config.command("model").description(t.config.modelDesc); + + model + .command("add") + .description(t.config.addDesc) + .requiredOption("--model-id ", t.config.addModelId) + .option("--provider ", t.config.addProvider) + .option("--api-key ", t.config.addApiKey) + .option("--base-url ", t.config.addBaseUrl) + .option("--context-window ", t.config.addContextWindow, parseIntArg) + .option("--client-type ", t.config.addClientType) + // Tri-state: --vision marks it supported / --no-vision marks it unsupported / neither given keeps the existing value (defaults to supported). + .option("--vision", t.config.addVision) + .option("--no-vision", t.config.addNoVision) + .option("--price-cache-read ", t.config.addPriceCacheRead, parseFloatArg) + .option("--price-cache-write ", t.config.addPriceCacheWrite, parseFloatArg) + .option("--price-output ", t.config.addPriceOutput, parseFloatArg) + .option("--project-id ", t.common.projectId, DEFAULT_PROJECT_ID) + .option("--set-default", t.config.addSetDefault, false) + .option("--root ", t.common.root) + .action(async (opts) => { + const root = resolveRootOption(opts.root); + // --model-id takes the upstream id, paired with --provider as a reference + // (--provider defaults to catalog-based inference, falling back to custom); no + // concatenation is performed. + const modelId: string = opts.modelId; + const provider: string = opts.provider ?? inferProviderForUpstream(modelId); + const ref: ModelRef = { provider, model_id: modelId }; + const before = await loadProjectConfig(root, opts.projectId); + const existed = getModel(before, ref) !== undefined; + // client_type default rule, only injected for new entries (updating an + // existing entry never overrides an explicit config): not set for first-party + // vendor groups (AgentHub auto-routes by upstream id, with env fallback keyed on + // id); defaults to openai for custom / self-hosted / gateway groups, with the + // gateway's endpoint base URL pre-filled as well. + const pInfo = providerInfo(provider); + const openAiDefault = + pInfo === undefined || pInfo.id === "custom" || pInfo.gatewayBaseUrl !== undefined; + const clientType: string | undefined = + opts.clientType ?? (!existed && openAiDefault ? "openai" : undefined); + const baseUrl: string | undefined = + opts.baseUrl ?? (!existed ? pInfo?.gatewayBaseUrl : undefined); + // Only collect explicitly given price fields, letting addModel merge them with the existing pricing per-field. + const pricing: Partial = {}; + if (opts.priceCacheRead !== undefined) pricing.cache_read = opts.priceCacheRead; + if (opts.priceCacheWrite !== undefined) pricing.cache_write = opts.priceCacheWrite; + if (opts.priceOutput !== undefined) pricing.output = opts.priceOutput; + const cfg = await addModel( + root, + opts.projectId, + { + provider, + model_id: modelId, + ...(opts.contextWindow !== undefined ? { context_window: opts.contextWindow } : {}), + ...(clientType !== undefined ? { client_type: clientType } : {}), + ...(opts.vision !== undefined ? { vision: opts.vision } : {}), + ...(Object.keys(pricing).length > 0 ? { pricing } : {}), + ...(opts.apiKey !== undefined ? { api_key: opts.apiKey } : {}), + ...(baseUrl !== undefined ? { base_url: baseUrl } : {}), + }, + { setDefault: Boolean(opts.setDefault) }, + ); + const defaultRef = cfg.default_model && formatModelRef(cfg.default_model); + const line = existed + ? t.modelUpdated(formatModelRef(ref), defaultRef) + : t.modelAdded(formatModelRef(ref), defaultRef); + process.stdout.write(`${line}\n`); + }); + + model + .command("default") + .description(t.config.defaultDesc) + .requiredOption("--model-id ", t.config.refModelId) + .requiredOption("--provider ", t.config.refProvider) + .option("--project-id ", t.common.projectId, DEFAULT_PROJECT_ID) + .option("--root ", t.common.root) + .action(async (opts) => { + const root = resolveRootOption(opts.root); + // --model-id takes the upstream id, paired with the required --provider as a + // reference (no concatenation, no fuzzy matching); setDefaultModel raises an error + // when the reference is not found in models. + const ref: ModelRef = { provider: opts.provider, model_id: opts.modelId }; + try { + await setDefaultModel(root, opts.projectId, ref); + } catch (err) { + process.stderr.write(`${t.error(err instanceof Error ? err.message : String(err))}\n`); + process.exitCode = 1; + return; + } + process.stdout.write(`${t.defaultModelSet(formatModelRef(ref))}\n`); + }); + + model + .command("vision") + .description(t.config.visionDesc) + .requiredOption("--model-id ", t.config.refModelId) + .requiredOption("--provider ", t.config.refProvider) + .option("--project-id ", t.common.projectId, DEFAULT_PROJECT_ID) + .option("--root ", t.common.root) + .action(async (opts) => { + const root = resolveRootOption(opts.root); + // Paired reference semantics match `model default`; existence and vision=false semantics validation is handled by setVisionModel. + const ref: ModelRef = { provider: opts.provider, model_id: opts.modelId }; + try { + await setVisionModel(root, opts.projectId, ref); + } catch (err) { + process.stderr.write(`${t.error(err instanceof Error ? err.message : String(err))}\n`); + process.exitCode = 1; + return; + } + process.stdout.write(`${t.visionModelSet(formatModelRef(ref))}\n`); + }); + + model + .command("list") + .description(t.config.listDesc) + .option("--project-id ", t.common.projectId, DEFAULT_PROJECT_ID) + .option("--root ", t.common.root) + .action(async (opts) => { + const root = resolveRootOption(opts.root); + const cfg = await loadProjectConfig(root, opts.projectId); + if (cfg.models.length === 0) { + process.stdout.write(`${t.modelListEmpty()}\n`); + return; + } + process.stdout.write(`${t.modelListTitle()}\n`); + for (const line of formatModelRows(cfg)) { + process.stdout.write(`${line}\n`); + } + }); + + const vault = config.command("vault").description(t.config.vaultDesc); + + vault + .command("set") + .description(t.config.vaultSetDesc) + .requiredOption("--key ", t.config.vaultKey) + .requiredOption("--value ", t.config.vaultValue) + .option("--project-id ", t.common.projectId, DEFAULT_PROJECT_ID) + .option("--agent-id ", t.common.agentId, DEFAULT_AGENT_ID) + .option("--root ", t.common.root) + .action(async (opts) => { + const root = resolveRootOption(opts.root); + try { + await setVaultEntry(root, opts.projectId, opts.agentId, opts.key, opts.value); + } catch (err) { + // Validation errors such as an invalid key name: print an explanation and exit with a non-zero code, without throwing a stack trace. + process.stderr.write(`${t.error(err instanceof Error ? err.message : String(err))}\n`); + process.exitCode = 1; + return; + } + process.stdout.write(`${t.vaultSet(opts.key)}\n`); + }); + + vault + .command("list") + .description(t.config.vaultListDesc) + .option("--project-id ", t.common.projectId, DEFAULT_PROJECT_ID) + .option("--agent-id ", t.common.agentId, DEFAULT_AGENT_ID) + .option("--root ", t.common.root) + .action(async (opts) => { + const root = resolveRootOption(opts.root); + const entries = Object.entries(await loadAgentVault(root, opts.projectId, opts.agentId)); + if (entries.length === 0) { + process.stdout.write(`${t.vaultListEmpty()}\n`); + return; + } + process.stdout.write(`${t.vaultListTitle()}\n`); + const width = Math.max(...entries.map(([key]) => key.length)); + for (const [key, value] of entries) { + process.stdout.write(`${key.padEnd(width)} ${maskApiKey(value)}\n`); + } + }); + + vault + .command("remove") + .description(t.config.vaultRemoveDesc) + .requiredOption("--key ", t.config.vaultKey) + .option("--project-id ", t.common.projectId, DEFAULT_PROJECT_ID) + .option("--agent-id ", t.common.agentId, DEFAULT_AGENT_ID) + .option("--root ", t.common.root) + .action(async (opts) => { + const root = resolveRootOption(opts.root); + const vaultEntries = await loadAgentVault(root, opts.projectId, opts.agentId); + if (vaultEntries[opts.key] === undefined) { + process.stderr.write(`${t.vaultKeyMissing(opts.key)}\n`); + process.exitCode = 1; + return; + } + await removeVaultEntry(root, opts.projectId, opts.agentId, opts.key); + process.stdout.write(`${t.vaultRemoved(opts.key)}\n`); + }); + + config + .command("lang") + .description(t.config.langDesc) + .argument("", t.config.langArg) + .action(async (language: string) => { + const lang = String(language).trim().toLowerCase(); + if (lang !== "zh" && lang !== "en") { + process.stderr.write(`${t.langInvalid(String(language))}\n`); + process.exitCode = 1; + return; + } + const { rcPath } = await applyLanguageToRc(lang, { + shell: process.env.SHELL, + home: homedir(), + }); + // The confirmation message is shown in the target language; the user must confirm before the shell restarts. + const m = getMessages(lang); + process.stdout.write(`${m.langSet(lang, rcPath)}\n`); + const interactive = Boolean(process.stdin.isTTY && process.stdout.isTTY); + if (interactive && (await confirmYes(m.langRestartConfirm()))) { + process.stdout.write(`${m.langRestart()}\n`); + restartShell(lang); + } else { + process.stdout.write(`${m.langRestartHint(rcPath)}\n`); + } + }); +} + +/** Interactive y/N confirmation; Ctrl-C (SIGINT) or input stream EOF/close are both treated as no, to avoid hanging. */ +function confirmYes(prompt: string): Promise { + const rl = createInterface({ input: process.stdin, output: process.stdout }); + return new Promise((resolve) => { + let done = false; + const finish = (value: boolean) => { + if (done) return; + done = true; + process.off("SIGINT", onSigint); + rl.close(); + resolve(value); + }; + const onSigint = () => finish(false); + process.once("SIGINT", onSigint); + rl.on("close", () => finish(false)); + rl.question(prompt, (answer) => finish(parseApprovalAnswer(answer) === "allow")); + }); +} + +function parseIntArg(value: string): number { + const n = Number.parseInt(value, 10); + if (Number.isNaN(n)) { + throw new Error(`无效的整数:${value}`); + } + return n; +} + +function parseFloatArg(value: string): number { + const n = Number.parseFloat(value); + if (Number.isNaN(n)) { + throw new Error(`无效的数值:${value}`); + } + return n; +} diff --git a/packages/cli/src/commands/run.ts b/packages/cli/src/commands/run.ts new file mode 100644 index 0000000..1a8f9d9 --- /dev/null +++ b/packages/cli/src/commands/run.ts @@ -0,0 +1,76 @@ +/** + * `penguin run` — send a single Task in one shot. + * + * penguin run -m [--model-id ] [--provider ] [--workspace ] + * [--project-id ] [--agent-id ] + * [--approve ] + * + * Uses the current directory when Workspace is unspecified; uses the Project's default model + * when model is unspecified. `--provider` is optional: when omitted, `--model-id` is resolved + * via resolveModelRef semantics (only matches when the exact value is globally unique in the + * config; ambiguity is an error). Defaults to interactive per-call approval; `--approve` + * selects the permission mode. + * Docs: /docs/cli § "penguin run". + */ +import type { Command } from "commander"; +import { createAgent, userText } from "@prismshadow/penguin-core"; +import { StreamRenderer } from "../render.js"; +import { runTask } from "../task-loop.js"; +import { denyActivePrompt, resolveApprovalMode } from "../approval.js"; +import type { Messages } from "../i18n.js"; + +export function registerRunCommand(program: Command, t: Messages): void { + program + .command("run") + .description(t.run.desc) + .requiredOption("-m, --message ", t.run.message) + .option("--model-id ", t.common.modelId) + .option("--provider ", t.common.provider) + .option("--project-id ", t.common.projectId) + .option("--agent-id ", t.common.agentId) + .option("--workspace ", t.common.workspace) + .option("--approve ", t.common.approve) + .action(async (opts) => { + const mode = resolveApprovalMode(opts.approve, t); + + const agent = await createAgent({ + ...(opts.agentId ? { agentId: opts.agentId } : {}), + ...(opts.projectId ? { projectId: opts.projectId } : {}), + }); + + const session = await agent.createSession({ + workspaceDir: opts.workspace ?? process.cwd(), + ...(opts.modelId ? { modelId: opts.modelId } : {}), + ...(opts.provider ? { provider: opts.provider } : {}), + }); + + const out = process.stdout; + out.write(`${t.header("run", agent.state.agentId, session.workspaceDir, session.modelId)}\n`); + + const controller = new AbortController(); + const onSigint = () => { + // Single SIGINT handler: Ctrl-C during approval collapses to "deny this tool" (see + // approval.ts); at all other times it interrupts the whole turn. + if (denyActivePrompt()) return; + controller.abort(); + }; + process.on("SIGINT", onSigint); + + const renderer = new StreamRenderer(out, t); + try { + const result = await runTask(session, [userText(opts.message)], { + mode, + signal: controller.signal, + renderer, + t, + }); + // Task ended with an abort (LLM failure/reconnect exhausted/user interrupt): non-zero + // exit code, for scripts/CI to check. + if (result.aborted) process.exitCode = 1; + } finally { + process.off("SIGINT", onSigint); + session.dispose(); // Tear down managed long-running command sessions to avoid leaking background processes + } + out.write("\n"); + }); +} diff --git a/packages/cli/src/commands/serve.ts b/packages/cli/src/commands/serve.ts new file mode 100644 index 0000000..67bc891 --- /dev/null +++ b/packages/cli/src/commands/serve.ts @@ -0,0 +1,133 @@ +/** + * `penguin server` / `penguin web` — starts the Web service. + * + * penguin server [--port ] [--host ] + * penguin web [--port ] [--host ] [--no-open] + * + * Both are entry points into the same service process: after setting PORT / HOST, it + * dynamically imports `@prismshadow/penguin-server` (whose entry point handles dotenv + * loading and graceful shutdown on its own), so the two never listen on separate ports + * in parallel. Port/host priority: command-line option > existing environment variable + * (including .env) > default 7364 / 127.0.0.1. `penguin web` additionally polls until the + * service is ready, prints the URL, and opens a browser per-platform (`--no-open` + * disables this). + * Docs: /docs/cli § "penguin server / penguin web". + */ +import { spawn } from "node:child_process"; +import type { Command } from "commander"; +import type { Messages } from "../i18n.js"; + +/** Default service port (deliberately avoids common defaults like 3000/8080). */ +export const DEFAULT_PORT = 7364; +/** Default service listen host. */ +export const DEFAULT_HOST = "127.0.0.1"; + +/** + * Resolves the listen port: command-line option takes priority, then the PORT + * environment variable, defaulting to 7364; throws if not an integer or out of the + * 0-65535 range. Exported for unit tests. + */ +export function resolvePort(option: string | undefined, env: string | undefined): number { + const raw = option ?? env; + if (raw === undefined || raw === "") return DEFAULT_PORT; + const port = Number(raw); + if (!Number.isInteger(port) || port < 0 || port > 65535) { + throw new Error(`Invalid port "${raw}". Use an integer between 0 and 65535.`); + } + return port; +} + +/** + * Picks the command to open a browser per-platform. On win32, `start` treats the first + * quoted argument as the window title, so an extra empty title placeholder is passed. + * Exported for unit tests. + */ +export function browserCommand(platform: string, url: string): { command: string; args: string[] } { + if (platform === "darwin") return { command: "open", args: [url] }; + if (platform === "win32") return { command: "cmd", args: ["/c", "start", "", url] }; + return { command: "xdg-open", args: [url] }; +} + +/** URL used for the readiness probe and browser access: when listening on a wildcard address (0.0.0.0 / ::), access via 127.0.0.1 instead. Exported for unit tests. */ +export function browserUrl(host: string, port: number): string { + const target = host === "0.0.0.0" || host === "::" ? "127.0.0.1" : host; + return `http://${target}:${port}/`; +} + +/** + * Sets PORT / HOST then starts the service: the server entry point only reads + * process.env, and its dotenv loading never overrides existing environment variables, + * so the values written here are the ones that take effect (options take priority over + * .env and any pre-existing env vars). + */ +async function startServer(opts: { + port?: string; + host?: string; +}): Promise<{ host: string; port: number }> { + const port = resolvePort(opts.port, process.env.PORT); + const host = opts.host ?? process.env.HOST ?? DEFAULT_HOST; + process.env.PORT = String(port); + process.env.HOST = host; + await import("@prismshadow/penguin-server"); + return { host, port }; +} + +/** Polls the service root path until it responds (any HTTP response counts as ready); keeps waiting on connection failure, returns false on timeout. */ +async function waitForReady(url: string, timeoutMs = 15_000, intervalMs = 300): Promise { + const deadline = Date.now() + timeoutMs; + for (;;) { + try { + // Each probe is capped at 1s: if the port is held by a non-HTTP program, the + // connection can succeed while the response hangs forever; without a timeout this + // would block the whole polling loop (the deadline check below would never run). + const res = await fetch(url, { signal: AbortSignal.timeout(1000) }); + void res.body?.cancel(); + return true; + } catch { + // The service isn't listening yet (or this probe timed out): keep polling. + } + if (Date.now() >= deadline) return false; + await new Promise((resolve) => setTimeout(resolve, intervalMs)); + } +} + +/** Opens the browser: spawn detached with output ignored; any failure is silently swallowed (failing to open doesn't affect the running service). */ +function openBrowser(url: string): void { + const { command, args } = browserCommand(process.platform, url); + try { + const child = spawn(command, args, { detached: true, stdio: "ignore" }); + child.on("error", () => {}); + child.unref(); + } catch { + // e.g. the browser command doesn't exist: ignore, the user can open it manually. + } +} + +export function registerServeCommands(program: Command, t: Messages): void { + program + .command("server") + .description(t.serve.serverDesc) + .option("--port ", t.serve.port) + .option("--host ", t.serve.host) + .action(async (opts: { port?: string; host?: string }) => { + await startServer(opts); + }); + + program + .command("web") + .description(t.serve.webDesc) + .option("--port ", t.serve.port) + .option("--host ", t.serve.host) + .option("--no-open", t.serve.noOpen) + .action(async (opts: { port?: string; host?: string; open: boolean }) => { + const { host, port } = await startServer(opts); + const url = browserUrl(host, port); + const ready = await waitForReady(url); + if (!ready) { + process.stdout.write(`${t.webTimeout(url)}\n`); + return; + } + process.stdout.write(`${t.webReady(url)}\n`); + if (opts.open) openBrowser(url); + }); +} diff --git a/packages/cli/src/i18n.ts b/packages/cli/src/i18n.ts new file mode 100644 index 0000000..c4dd048 --- /dev/null +++ b/packages/cli/src/i18n.ts @@ -0,0 +1,390 @@ +/** + * CLI text internationalization (i18n). + * + * Language comes from the `PENGUIN_LANG` env var (`en` / `zh`), defaulting to English (en) — + * independent of Project config or CLI options. This module centralizes all user-visible text: + * command/option help descriptions and runtime output, one implementation per language. + */ + +/** UI language. */ +export type Language = "en" | "zh"; + +/** Resolve the language from the env var; `zh` matches exactly, everything else falls back to English (see comment #2). */ +export function resolveLanguage(): Language { + const v = (process.env.PENGUIN_LANG ?? "").trim().toLowerCase(); + return v === "zh" ? "zh" : "en"; +} + +export interface Messages { + // —— Command/option help descriptions —— + cliDescription: string; + versionDesc: string; + common: { + projectId: string; + agentId: string; + modelId: string; + /** run/chat's --provider: pairs with --model-id; when omitted, resolved by unique match (ambiguity is an error). */ + provider: string; + /** Data root directory option (priority: --root > PENGUIN_HOME > ~/.penguin/data). */ + root: string; + workspace: string; + approve: string; + }; + config: { + desc: string; + modelDesc: string; + addDesc: string; + addModelId: string; + addProvider: string; + addApiKey: string; + addBaseUrl: string; + addContextWindow: string; + addClientType: string; + addVision: string; + addNoVision: string; + addPriceCacheRead: string; + addPriceCacheWrite: string; + addPriceOutput: string; + addSetDefault: string; + defaultDesc: string; + visionDesc: string; + /** `model default` / `model vision`'s --model-id: the upstream request id (pairs with --provider as a reference). */ + refModelId: string; + /** `model default` / `model vision`'s --provider: the provider group of the referenced entry (required). */ + refProvider: string; + listDesc: string; + langDesc: string; + langArg: string; + vaultDesc: string; + vaultSetDesc: string; + vaultListDesc: string; + vaultRemoveDesc: string; + vaultKey: string; + vaultValue: string; + }; + run: { desc: string; message: string }; + chat: { desc: string; resume: string }; + serve: { + serverDesc: string; + webDesc: string; + port: string; + host: string; + noOpen: string; + }; + + // —— Runtime output —— + header(kind: "chat" | "run", agentId: string, workspace: string, model: string): string; + chatHints(): string; + confirmExit(): string; + taskInterrupted(): string; + error(message: string): string; + /** Approval prompt text (the tool call is already streamed above and directly precedes this prompt, so no index and no re-rendering). */ + approvePrompt(): string; + /** + * Stats shown at the end of each Task: Session cumulative values plus this task's delta — + * context window length, Token usage, elapsed time. Delta strings carry their own sign + * (contextDelta can be negative after context is compacted), e.g. + * `[stats] context 4k (+1k) · tokens 6k (+1.2k) · 5.1s (+2.3s)`. + */ + taskStats(s: { + context: string; + contextDelta: string; + tokens: string; + tokensDelta: string; + elapsed: string; + elapsedDelta: string; + }): string; + /** Abort event label (may include a reason). */ + abortLabel(reason?: string): string; + /** request_end ended with timeout/malformed: the engine retries (reconnect) carrying already-produced content; attempt is the retry count. */ + reconnectLabel(status: "timeout" | "malformed", attempt: number): string; + /** compaction start event: indicates compaction in progress (mode is summarize/discard, reason is context/turns/manual). */ + compactionStart(mode: string, reason: string): string; + /** + * compaction stop event: the compaction result (status is completed/failed/aborted; + * completed varies its text by mode). tokens is Token usage (same convention as the stats + * line: total = Session cumulative, delta = consumed by this compaction, carrying its own + * sign); when present it is appended at the end of the line, e.g. ` · tokens 14k (+6k)`. + */ + compactionStop(mode: string, status: string, tokens?: { total: string; delta: string }): string; + /** Prompt shown when `/compact` has nothing to compact (session just started / two consecutive compactions). */ + compactNothing(): string; + /** Prompt for an invalid --approve mode. */ + approveModeInvalid(value: string): string; + /** Render label for an approval decision (frontend renders the approval_decision event; one label each for allow/deny). */ + approvalDecision(decision: "allow" | "deny"): string; + /** --resume is mutually exclusive with --workspace/--model-id (neither can change once the Session is created). */ + resumeNoOverride(): string; + /** --resume given without a session id, and the current Agent has no Session at all. */ + resumeNoSession(): string; + /** One-line prompt shown after a successful resume, before rendering history. */ + resumedBanner(sessionId: string, messageCount: number): string; + /** Example resume command shown when the REPL exits (dim print; only when this session has a resumable record). */ + resumeHint(command: string): string; + langInvalid(value: string): string; + langSet(lang: string, rcPath: string): string; + langRestartConfirm(): string; + langRestart(): string; + langRestartHint(rcPath: string): string; + /** Result output for model add/default/vision: the argument is the already-formatted pair reference (formatModelRef). */ + modelAdded(model: string, defaultModel: string | undefined): string; + modelUpdated(model: string, defaultModel: string | undefined): string; + defaultModelSet(model: string): string; + visionModelSet(model: string): string; + modelListTitle(): string; + modelListEmpty(): string; + vaultSet(key: string): string; + vaultRemoved(key: string): string; + vaultKeyMissing(key: string): string; + vaultListTitle(): string; + vaultListEmpty(): string; + /** URL prompt once the `penguin web` service is ready. */ + webReady(url: string): string; + /** Manual-open prompt after the `penguin web` ready-poll times out (15s). */ + webTimeout(url: string): string; +} + +function header(kind: "chat" | "run", agentId: string, workspace: string, model: string): string { + return `PenguinHarness ${kind} — agent=${agentId} workspace=${workspace} model=${model}`; +} + +const en: Messages = { + cliDescription: "PenguinHarness CLI", + versionDesc: "output the version number", + common: { + projectId: "Project id", + agentId: "Agent id", + modelId: "Model to use (upstream model id; defaults to the Project default model)", + provider: + "Provider of --model-id; when omitted, the model id must match exactly one configured entry (ambiguity is an error)", + root: "Data root directory (overrides PENGUIN_HOME and ~/.penguin/data)", + workspace: "Workspace directory; must already exist (defaults to the current directory)", + approve: + "Approval mode: allow-all (auto-approve, default), deny-all (auto-reject), read-only (auto-approve read-only tools, prompt for the rest), always-ask (prompt per tool)", + }, + config: { + desc: "Manage Project configuration", + modelDesc: "Manage model credentials and the default model", + addDesc: "Add or update a model, optionally writing a credential", + addModelId: "Upstream model id sent to AgentHub as-is (e.g. claude-sonnet-4-6)", + addProvider: + "Provider group stored alongside model_id; inferred from the builtin catalog when omitted, else custom", + addApiKey: "API key, stored inline in the Project's hidden .project_config.toml", + addBaseUrl: "Custom base URL", + addContextWindow: "Context window size (tokens)", + addClientType: "AgentHub client type (e.g. openai); inferred from model id when omitted", + addVision: "Mark the model as supporting image input (vision)", + addNoVision: "Mark the model as NOT supporting image input; omit both to keep current", + addPriceCacheRead: "Price per 1M tokens: cache read (USD)", + addPriceCacheWrite: "Price per 1M tokens: cache write (USD)", + addPriceOutput: "Price per 1M tokens: output (USD)", + addSetDefault: "Also set as the Project default model", + defaultDesc: "Set the Project default model", + visionDesc: "Set the vision model used by read_image for non-vision session models", + refModelId: "Upstream model id; forms the (provider, model_id) pair reference with --provider", + refProvider: "Provider group of the referenced entry (see `penguin config model list`)", + listDesc: "List the Project's models (API keys hidden)", + langDesc: + "Set the interface language (en|zh); persists PENGUIN_LANG to your shell startup file", + langArg: "Language: en or zh", + vaultDesc: "Manage an Agent's vault (environment variables injected into its shell commands)", + vaultSetDesc: "Set a vault environment variable (added or overwritten)", + vaultListDesc: "List vault environment variables (values masked)", + vaultRemoveDesc: "Remove a vault environment variable", + vaultKey: "Variable name (letters, digits and underscores; must not start with a digit)", + vaultValue: "Variable value, written to the Agent's agent_state/.vault.toml", + }, + run: { desc: "Run a single Task", message: "Prompt for this Task" }, + chat: { + desc: "Open the interactive REPL", + resume: + "Resume an existing Session (defaults to the agent's most recent one); workspace and model follow the original Session", + }, + serve: { + serverDesc: "Start the Web service (HTTP API and the built-in frontend, same process)", + webDesc: "Start the Web service and open the UI in a browser once it is ready", + port: "Listen port (falls back to the PORT env var, default 7364)", + host: "Listen address (falls back to the HOST env var, default 127.0.0.1)", + noOpen: "Do not open a browser automatically", + }, + + header, + chatHints: () => + "Type a message to start a conversation; end a line with \\; /compact to compact the context; /exit to quit; and Ctrl-C interrupts the current conversation.", + confirmExit: () => "Exit penguin? [y/N] ", + taskInterrupted: () => "[current conversation interrupted]", + error: (message) => `[error] ${message}`, + approvePrompt: () => "? Approve this tool call? [Y/n] ", + taskStats: (s) => + `[stats] context ${s.context} (${s.contextDelta}) · tokens ${s.tokens} (${s.tokensDelta}) · ${s.elapsed} (${s.elapsedDelta})`, + abortLabel: (reason) => `[abort]${reason ? `: ${reason}` : ""}`, + reconnectLabel: (status, attempt) => + `[retry] ${status === "timeout" ? "connection timed out" : "response incomplete or unparseable"}; sending retry #${attempt}…`, + compactionStart: (mode, reason) => + mode === "discard" + ? `[compaction] discarding context (${reason})…` + : `[compaction] summarizing context (${reason})…`, + compactionStop: (mode, status, tokens) => + (status === "completed" + ? mode === "discard" + ? "[compaction] done; old context discarded" + : "[compaction] done; continuing with the summarized context" + : `[compaction] ${status}; keeping the current context`) + + (tokens ? ` · tokens ${tokens.total} (${tokens.delta})` : ""), + compactNothing: () => "[compaction] nothing to compact yet", + approveModeInvalid: (value) => + `Invalid approval mode "${value}". Use allow-all, deny-all, read-only, or always-ask.`, + approvalDecision: (decision) => (decision === "allow" ? "✓ [approved]" : "× [denied]"), + resumeNoOverride: () => + "--resume does not accept --workspace, --model-id or --provider: they follow the original Session and cannot change.", + resumeNoSession: () => "No session to resume: this agent has no recorded sessions yet.", + resumedBanner: (sessionId, messageCount) => + `[resumed] ${sessionId} · ${messageCount} message${messageCount === 1 ? "" : "s"} in the current context`, + resumeHint: (command) => `To continue this conversation: ${command}`, + langInvalid: (value) => `Invalid language "${value}". Use en or zh.`, + langSet: (lang, rcPath) => `Language set to ${lang}; wrote PENGUIN_LANG to ${rcPath}.`, + langRestartConfirm: () => "Open a new shell now to apply? [y/N] ", + langRestart: () => "Opening a new shell with the new language (type exit to return)…", + langRestartHint: (rcPath) => `Open a new terminal, or run: source ${rcPath}`, + modelAdded: (model, def) => `Added model ${model}. Default model: ${def ?? "(unset)"}`, + modelUpdated: (model, def) => `Updated model ${model}. Default model: ${def ?? "(unset)"}`, + defaultModelSet: (model) => `Default model set to ${model}.`, + visionModelSet: (model) => `Vision model set to ${model}.`, + modelListTitle: () => "Configured models:", + modelListEmpty: () => "No models configured yet. Add one with `penguin config model add`.", + vaultSet: (key) => `Saved vault entry ${key}.`, + vaultRemoved: (key) => `Removed vault entry ${key}.`, + vaultKeyMissing: (key) => `Vault entry ${key} does not exist.`, + vaultListTitle: () => "Vault environment variables (values masked):", + vaultListEmpty: () => "The vault is empty. Add one with `penguin config vault set`.", + webReady: (url) => `Web UI ready: ${url}`, + webTimeout: (url) => `Server is not responding yet; open ${url} manually once it is ready.`, +}; + +const zh: Messages = { + cliDescription: "PenguinHarness CLI", + versionDesc: "输出版本号", + common: { + projectId: "Project id", + agentId: "Agent id", + modelId: "本次使用的模型(上游模型 id;默认 Project 默认模型)", + provider: "--model-id 的 provider 分组;省略时 model id 须在配置中精确唯一命中(歧义报错)", + root: "数据根目录(优先于 PENGUIN_HOME 与 ~/.penguin/data)", + workspace: "Workspace 目录,须为已存在目录(默认当前目录)", + approve: + "审批模式:allow-all(全部放行,缺省)、deny-all(全部拒绝)、read-only(自动放行只读工具,其余仍逐个询问)、always-ask(逐个询问)", + }, + config: { + desc: "管理 Project 配置", + modelDesc: "管理模型 credential 与默认模型", + addDesc: "新增或更新一个模型,并可写入 credential", + addModelId: "上游模型 id(如 claude-sonnet-4-6,原样发给 AgentHub)", + addProvider: "与 model_id 分列存储的 provider 分组;缺省按内置目录推断,推断不出为 custom", + addApiKey: "API key,内联存入 Project 的隐藏文件 .project_config.toml", + addBaseUrl: "自定义 base url", + addContextWindow: "上下文窗口大小(token 数)", + addClientType: "AgentHub 客户端协议(如 openai);缺省由 model id 推断", + addVision: "标注该模型支持图片输入(视觉)", + addNoVision: "标注该模型不支持图片输入;两者都不给则保留原值", + addPriceCacheRead: "每百万 token 价格:缓存读取(USD)", + addPriceCacheWrite: "每百万 token 价格:缓存写入(USD)", + addPriceOutput: "每百万 token 价格:输出(USD)", + addSetDefault: "同时设为该 Project 的默认模型", + defaultDesc: "设置 Project 的默认模型", + visionDesc: "设置 read_image 代读用的视觉模型(供不支持图片的会话模型读图)", + refModelId: "上游模型 id;与 --provider 构成 (provider, model_id) 成对引用", + refProvider: "引用条目的 provider 分组(见 `penguin config model list`)", + listDesc: "列出当前 Project 的模型(API key 隐藏)", + langDesc: "设置界面语言(en|zh);将 PENGUIN_LANG 写入 shell 启动文件并持久化", + langArg: "语言:en 或 zh", + vaultDesc: "管理 Agent vault(注入该 Agent shell 命令的环境变量)", + vaultSetDesc: "写入一个 vault 环境变量(不存在则新增,存在则覆盖)", + vaultListDesc: "列出 vault 环境变量(值掩码显示)", + vaultRemoveDesc: "删除一个 vault 环境变量", + vaultKey: "变量名(字母、数字与下划线,不能以数字开头)", + vaultValue: "变量值,写入该 Agent 的 agent_state/.vault.toml", + }, + run: { desc: "单次运行一个 Task", message: "本次 Task 的 Prompt" }, + chat: { + desc: "打开交互式 REPL", + resume: + "恢复既有 Session 继续对话(缺省恢复当前 Agent 最近一次);Workspace 与模型沿用原 Session", + }, + serve: { + serverDesc: "启动 Web 服务(HTTP API 与内置前端,同一进程)", + webDesc: "启动 Web 服务,就绪后用浏览器打开界面", + port: "监听端口(其次取环境变量 PORT,缺省 7364)", + host: "监听地址(其次取环境变量 HOST,缺省 127.0.0.1)", + noOpen: "不自动打开浏览器", + }, + + header, + chatHints: () => + "输入消息发起对话;行尾 \\ 续行;/compact 压缩上下文;/exit 退出;Ctrl-C 中断对话。", + confirmExit: () => "确认退出 penguin?[y/N] ", + taskInterrupted: () => "[已中断当前对话]", + error: (message) => `[错误] ${message}`, + approvePrompt: () => "? 批准此工具调用?[Y/n] ", + taskStats: (s) => + `[统计信息] 上下文 ${s.context} (${s.contextDelta}) · tokens ${s.tokens} (${s.tokensDelta}) · 用时 ${s.elapsed} (${s.elapsedDelta})`, + abortLabel: (reason) => `[已中断]${reason ? `:${reason}` : ""}`, + reconnectLabel: (status, attempt) => + `[重试] ${status === "timeout" ? "连接超时或网络中断" : "响应不完整或无法解析"},正在发起第 ${attempt} 次重试……`, + compactionStart: (mode, reason) => + mode === "discard" + ? `[压缩] 正在丢弃旧上下文(${reason})……` + : `[压缩] 正在总结压缩上下文(${reason})……`, + compactionStop: (mode, status, tokens) => + (status === "completed" + ? mode === "discard" + ? "[压缩] 完成,旧上下文已丢弃" + : "[压缩] 完成,已切换到摘要后的新上下文" + : `[压缩] ${status === "aborted" ? "已中断" : "失败"},保留当前上下文`) + + (tokens ? ` · tokens ${tokens.total} (${tokens.delta})` : ""), + compactNothing: () => "[压缩] 当前上下文为空,无需压缩", + approveModeInvalid: (value) => + `无效的审批模式 "${value}"。请使用 allow-all、deny-all、read-only 或 always-ask。`, + approvalDecision: (decision) => (decision === "allow" ? "✓ [已批准]" : "× [已拒绝]"), + resumeNoOverride: () => + "--resume 不接受 --workspace、--model-id 与 --provider:均沿用原 Session,创建后不可更换。", + resumeNoSession: () => "没有可恢复的 Session:当前 Agent 还没有任何会话记录。", + resumedBanner: (sessionId, messageCount) => + `[已恢复] ${sessionId} · 当前上下文共 ${messageCount} 条消息`, + resumeHint: (command) => `继续本次对话:${command}`, + langInvalid: (value) => `无效的语言 "${value}"。请使用 en 或 zh。`, + langSet: (lang, rcPath) => `语言已设为 ${lang};已将 PENGUIN_LANG 写入 ${rcPath}。`, + langRestartConfirm: () => "现在打开新 shell 使其生效?[y/N] ", + langRestart: () => "正在打开使用新语言的新 shell(输入 exit 可返回)……", + langRestartHint: (rcPath) => `请打开新终端,或执行:source ${rcPath}`, + modelAdded: (model, def) => `已添加模型 ${model}。当前默认模型:${def ?? "(未设置)"}`, + modelUpdated: (model, def) => `已更新模型 ${model}。当前默认模型:${def ?? "(未设置)"}`, + defaultModelSet: (model) => `默认模型已设为 ${model}。`, + visionModelSet: (model) => `视觉模型已设为 ${model}。`, + modelListTitle: () => "已配置的模型:", + modelListEmpty: () => "尚未配置任何模型。用 `penguin config model add` 添加。", + vaultSet: (key) => `已保存 vault 条目 ${key}。`, + vaultRemoved: (key) => `已删除 vault 条目 ${key}。`, + vaultKeyMissing: (key) => `vault 条目 ${key} 不存在。`, + vaultListTitle: () => "vault 环境变量(值已掩码):", + vaultListEmpty: () => "vault 为空。用 `penguin config vault set` 添加。", + webReady: (url) => `Web 界面已就绪:${url}`, + webTimeout: (url) => `服务尚未就绪,请稍后手动打开 ${url}。`, +}; + +/** Get the message set for a language. */ +export function getMessages(language: Language): Messages { + return language === "zh" ? zh : en; +} + +/** Resolve the language from the env var and return its message set (the default used when no explicit `t` is given). */ +export function defaultMessages(): Messages { + return getMessages(resolveLanguage()); +} + +/** Mask an API key: keep only a few trailing characters; return `-` when unconfigured. */ +export function maskApiKey(apiKey: string | undefined): string { + if (!apiKey) return "-"; + // Mask the whole thing when ≤12 chars: `****last4` reveals too much of a short secret (same threshold as the server-side mask). + if (apiKey.length <= 12) return "***"; + return `****${apiKey.slice(-4)}`; +} diff --git a/packages/cli/src/index.ts b/packages/cli/src/index.ts new file mode 100644 index 0000000..de396e0 --- /dev/null +++ b/packages/cli/src/index.ts @@ -0,0 +1,46 @@ +/** + * PenguinHarness CLI entry point. + * + * Only responsible for parsing CLI input into SDK arguments and rendering the streaming + * OmniMessage returned by the SDK. + * Loads .env on startup (e.g. locally configured ANTHROPIC_API_KEY / ANTHROPIC_BASE_URL). + * + * penguin config model add|default ... + * penguin chat ... + * penguin run --message ... + * penguin server|web ... + * Docs: packages/docs/content/cli.{zh,en}.md (site path /docs/cli). + */ +import "dotenv/config"; +import { Command } from "commander"; +import { VERSION } from "@prismshadow/penguin-core"; +import { registerConfigCommand } from "./commands/config.js"; +import { registerRunCommand } from "./commands/run.js"; +import { registerChatCommand } from "./commands/chat.js"; +import { registerServeCommands } from "./commands/serve.js"; +import { defaultMessages } from "./i18n.js"; + +// Language comes from the PENGUIN_LANG env var (default en); used consistently for +// command/option descriptions and runtime output. +const t = defaultMessages(); + +const program = new Command(); +program + .name("penguin") + .description(t.cliDescription) + .version(VERSION, "-v, --version", t.versionDesc); + +registerConfigCommand(program, t); +registerRunCommand(program, t); +registerChatCommand(program, t); +registerServeCommands(program, t); + +// Show help only when no subcommand is given (empty input); do not error. +program.action(() => { + program.outputHelp(); +}); + +program.parseAsync(process.argv).catch((err: unknown) => { + process.stderr.write(`${err instanceof Error ? err.message : String(err)}\n`); + process.exitCode = 1; +}); diff --git a/packages/cli/src/input.ts b/packages/cli/src/input.ts new file mode 100644 index 0000000..c609b9a --- /dev/null +++ b/packages/cli/src/input.ts @@ -0,0 +1,121 @@ +/** + * CLI input-layer helpers: multi-line input and paste support. + * + * - `PasteFilter`: a Transform inserted between stdin and readline. Once terminal bracketed + * paste mode is enabled, pasted content is wrapped in `\x1b[200~` … `\x1b[201~`; this + * Transform strips that pair of markers, withholds the pasted content in between (not + * forwarded to readline, so internal newlines aren't split into multiple submissions), and + * emits it as a whole via a `paste` event. All other keystrokes are forwarded to readline + * unchanged, preserving line editing and Ctrl-C. + * - `LineComposer`: assembles "line-by-line input + paste blocks" into one complete message. + * A single trailing backslash `\` means line continuation; a paste block goes into the + * pending buffer as a whole and is sent on Enter. + */ +import { Transform, type TransformCallback } from "node:stream"; + +const PASTE_START = "\x1b[200~"; +const PASTE_END = "\x1b[201~"; + +/** + * Return the trailing part of `data` that could be a prefix of `marker` (hold, kept for + * concatenation with the next chunk); the rest is ready to process immediately (emit). Handles + * the case where a marker straddles a data-chunk boundary. + */ +export function splitTrailingPartial(data: string, marker: string): { emit: string; hold: string } { + const max = Math.min(marker.length - 1, data.length); + for (let k = max; k > 0; k--) { + if (data.endsWith(marker.slice(0, k))) { + return { emit: data.slice(0, data.length - k), hold: data.slice(data.length - k) }; + } + } + return { emit: data, hold: "" }; +} + +export class PasteFilter extends Transform { + private inPaste = false; + private pasteBuf = ""; + private leftover = ""; + + override _transform(chunk: Buffer | string, _enc: BufferEncoding, cb: TransformCallback): void { + let data = this.leftover + chunk.toString("utf8"); + this.leftover = ""; + + while (data.length > 0) { + if (!this.inPaste) { + const i = data.indexOf(PASTE_START); + if (i === -1) { + const { emit, hold } = splitTrailingPartial(data, PASTE_START); + if (emit) this.push(emit); + this.leftover = hold; + data = ""; + } else { + if (i > 0) this.push(data.slice(0, i)); + data = data.slice(i + PASTE_START.length); + this.inPaste = true; + this.pasteBuf = ""; + } + } else { + const j = data.indexOf(PASTE_END); + if (j === -1) { + const { emit, hold } = splitTrailingPartial(data, PASTE_END); + this.pasteBuf += emit; + this.leftover = hold; + data = ""; + } else { + this.pasteBuf += data.slice(0, j); + data = data.slice(j + PASTE_END.length); + this.inPaste = false; + const text = this.pasteBuf; + this.pasteBuf = ""; + this.emit("paste", text); + } + } + } + cb(); + } +} + +/** Whether the line ends in a continuation (an odd number of trailing backslashes; an even count is treated as escaped literal backslashes). */ +export function endsWithContinuation(line: string): boolean { + const trailing = line.match(/(\\+)$/)?.[1] ?? ""; + return trailing.length % 2 === 1; +} + +/** + * Assembles line-by-line input and paste blocks into a complete message. + * `pushTypedLine` returns `{ message }` when a message is ready, or `{}` while still + * continuing/pending. + */ +export class LineComposer { + private pending: string[] = []; + + pushTypedLine(line: string): { message?: string } { + if (endsWithContinuation(line)) { + this.pending.push(line.slice(0, -1)); + return {}; + } + if (this.pending.length > 0) { + const lines = line === "" ? this.pending : [...this.pending, line]; + this.pending = []; + return { message: lines.join("\n") }; + } + return { message: line }; + } + + /** Accept a paste block (strip trailing blank lines, normalize newlines); it goes into the pending buffer as a whole, waiting to be sent on Enter. */ + pushPaste(text: string): { lineCount: number; normalized: string } { + const norm = text.replace(/\r\n?/g, "\n").replace(/\n+$/, ""); + if (norm.length === 0) return { lineCount: 0, normalized: "" }; + const lines = norm.split("\n"); + this.pending.push(...lines); + return { lineCount: lines.length, normalized: norm }; + } + + hasPending(): boolean { + return this.pending.length > 0; + } + + reset(): void { + this.pending = []; + } +} diff --git a/packages/cli/src/lang-config.ts b/packages/cli/src/lang-config.ts new file mode 100644 index 0000000..6c156d2 --- /dev/null +++ b/packages/cli/src/lang-config.ts @@ -0,0 +1,102 @@ +/** + * Language persistence: write `PENGUIN_LANG` into the user's shell startup file, then restart + * the shell so it takes effect. + * + * A child process can't modify its parent shell's environment variables directly, so + * `penguin config lang` uses a "write the startup file + restart the shell" approach: write + * `export PENGUIN_LANG=` into the shell startup file inside a marked block (idempotent, + * updates in place), then open an interactive shell carrying the new language env var. New + * terminals will read the variable from the startup file, so it persists. + */ +import { spawn } from "node:child_process"; +import { mkdir, readFile, writeFile } from "node:fs/promises"; +import { dirname, join } from "node:path"; +import type { Language } from "./i18n.js"; + +const BEGIN = "# >>> PenguinHarness PENGUIN_LANG >>>"; +const END = "# <<< PenguinHarness PENGUIN_LANG <<<"; + +export type ShellKind = "zsh" | "bash" | "fish" | "unknown"; + +export interface ShellRc { + kind: ShellKind; + /** Absolute path to the startup file. */ + rcPath: string; + /** Generate the export line for a given language (shell-syntax specific). */ + body(lang: Language): string; +} + +/** Resolve the startup file and export syntax from `$SHELL`. Falls back to `~/.profile` for an unknown shell. */ +export function resolveShellRc(shell: string | undefined, home: string): ShellRc { + const base = (shell ?? "").split("/").pop()?.toLowerCase() ?? ""; + if (base.includes("fish")) { + return { + kind: "fish", + rcPath: join(home, ".config", "fish", "config.fish"), + body: (lang) => `set -gx PENGUIN_LANG ${lang}`, + }; + } + if (base.includes("zsh")) { + return { + kind: "zsh", + rcPath: join(home, ".zshrc"), + body: (lang) => `export PENGUIN_LANG=${lang}`, + }; + } + if (base.includes("bash")) { + return { + kind: "bash", + rcPath: join(home, ".bashrc"), + body: (lang) => `export PENGUIN_LANG=${lang}`, + }; + } + return { + kind: "unknown", + rcPath: join(home, ".profile"), + body: (lang) => `export PENGUIN_LANG=${lang}`, + }; +} + +/** Insert or update the marked PenguinHarness block in place within the text; leaves the rest of the content unchanged. */ +export function upsertBlock(content: string, bodyLine: string): string { + const block = `${BEGIN}\n${bodyLine}\n${END}`; + const begin = content.indexOf(BEGIN); + const end = content.indexOf(END); + if (begin !== -1 && end !== -1 && end > begin) { + const before = content.slice(0, begin); + const after = content.slice(end + END.length); + return `${before}${block}${after}`; + } + // Append at the end: leave a blank line before it if there's existing content. + if (content.length === 0) return `${block}\n`; + const sep = content.endsWith("\n") ? "" : "\n"; + return `${content}${sep}\n${block}\n`; +} + +/** Write the language into the shell startup file (creating the directory if needed). Returns the file path written and the shell kind. */ +export async function applyLanguageToRc( + lang: Language, + opts: { shell: string | undefined; home: string }, +): Promise<{ rcPath: string; kind: ShellKind }> { + const rc = resolveShellRc(opts.shell, opts.home); + await mkdir(dirname(rc.rcPath), { recursive: true }); + let content = ""; + try { + content = await readFile(rc.rcPath, "utf8"); + } catch { + /* File doesn't exist yet; treat as empty content */ + } + await writeFile(rc.rcPath, upsertBlock(content, rc.body(lang)), "utf8"); + return { rcPath: rc.rcPath, kind: rc.kind }; +} + +/** Open an interactive shell carrying the new language env var; this process exits when the user exits that shell. */ +export function restartShell(lang: Language): void { + const shell = process.env.SHELL || "/bin/zsh"; + const child = spawn(shell, ["-i"], { + stdio: "inherit", + env: { ...process.env, PENGUIN_LANG: lang }, + }); + child.on("exit", (code) => process.exit(code ?? 0)); + child.on("error", () => process.exit(1)); +} diff --git a/packages/cli/src/render.ts b/packages/cli/src/render.ts new file mode 100644 index 0000000..b8b06b8 --- /dev/null +++ b/packages/cli/src/render.ts @@ -0,0 +1,874 @@ +/** + * CLI streaming renderer. + * + * Rendering rule: **only the streaming `partial_*` variants of model_msg are rendered**; + * complete (non-streaming) model_msg is never rendered. A complete message's content has + * already been delivered by its corresponding `partial_*` stream, so re-rendering it would + * be redundant. `partial_*` is written out token by token as it arrives. + * event_msg is not message rendering and is handled separately: `token_usage` accumulates + * and is summarized in the `[stats]` line at task end, `approval_decision` prints one line + * with the approval result, `abort` prints one line noting the interruption, and each of + * `compaction_begin`/`compaction_end` prints one line of compaction progress; + * `session_meta` is never rendered. + * + * **Screen lock (concurrent tools)**: tools run concurrently and asynchronously, so + * messages may arrive interleaved. The renderer queues internally to guarantee: + * - a streaming segment (the LLM's text/thinking/tool_call stream, or a given tool's + * output stream start->delta->stop) holds the screen until stop, while other messages + * queue up; + * - all output is locked while waiting for user input (the approval prompt, + * `beginUserPrompt`/`endUserPrompt`); + * - when the head of the queue is held, the holder's own subsequent messages are let + * through first (preserving in-segment order), avoiding deadlock. + * + * **Pairing tags**: a tool call and its output may be separated by several segments, so + * both are tagged with a shared word for pairing: the call line reads + * `[tool-653] $ cmd`, the output line `[tool-653] >> ...` (653 being the last 3 + * characters of tool_call_id); nested (subagent) tools use + * `[agent-f2a-tool-653] $ cmd` (f2a being the last 3 characters of the direct child + * Session id). Approval lines carry no tag (they immediately follow the matching call + * line, so context makes the pairing clear): `[approved]`. + * + * **Nested sub-session messages** (those carrying an origin) are handled separately: + * child tool calls (so the user can see what the subagent is calling before approval) + * and child approval results are rendered, and child token_usage counts toward this + * task's delta and the Session total; everything else (child text/thinking, etc.) is + * not rendered — the child Agent's final text is already streamed through the parent + * tool's output gutter. + * + * No third-party color library is used; only minimal ANSI escapes. + */ +import { isEventMessage, isModelMessage } from "@prismshadow/penguin-core"; +import type { + AbortPayload, + ApprovalDecision, + ApprovalDecisionPayload, + CompactionBeginPayload, + CompactionEndPayload, + MessageOrigin, + OmniMessage, + PartialTextPayload, + PartialThinkingPayload, + PartialToolCallPayload, + PartialToolCallOutputPayload, + RequestEndPayload, + TokenUsagePayload, + ToolCallPayload, +} from "@prismshadow/penguin-core"; +import { renderPartialToolCall } from "./tool-render.js"; +import { defaultMessages } from "./i18n.js"; +import type { Messages } from "./i18n.js"; + +const DIM = "\x1b[2m"; +const CYAN = "\x1b[36m"; +const RESET = "\x1b[0m"; + +export function dim(text: string): string { + return `${DIM}${text}${RESET}`; +} + +/** Colors a tool call line cyan, distinguishing it from body text/thinking (review comment #5). */ +function cyan(text: string): string { + return `${CYAN}${text}${RESET}`; +} + +/** Takes the last 3 characters of an id as the on-screen pairing number. */ +function shortId(id: string): string { + return id.slice(-3); +} + +/** + * On-screen pairing tag for a tool call/output: main-session tools -> + * `tool-`; nested (subagent) tools -> + * `agent--tool-`. + */ +function callTag(toolCallId: string, origin?: readonly MessageOrigin[]): string { + const tid = `tool-${shortId(toolCallId)}`; + return origin && origin.length > 0 ? `agent-${shortId(origin[origin.length - 1]!)}-${tid}` : tid; +} + +/** Converts a token count to a human-readable abbreviation: 1234->1.2k, 1500000->1.5M, <1000 unchanged. */ +export function humanizeTokens(n: number): string { + const abs = Math.abs(n); + if (abs < 1000) return `${n}`; + if (abs < 1_000_000) { + const v = n / 1000; + return `${trimZero(v)}k`; + } + const v = n / 1_000_000; + return `${trimZero(v)}M`; +} + +/** Keeps one decimal place but drops a trailing `.0`. */ +function trimZero(v: number): string { + const s = v.toFixed(1); + return s.endsWith(".0") ? s.slice(0, -2) : s; +} + +/** Adds an explicit sign to a delta string: non-negative gets a `+` prefix, negative already has its own `-` (context can go negative after compaction shrinks it). */ +function signedDelta(formatted: string): string { + return formatted.startsWith("-") ? formatted : `+${formatted}`; +} + +/** Converts milliseconds into a human-readable duration: `820ms`, `2.3s`, `1m3s`. */ +function humanizeDuration(ms: number): string { + if (ms < 1000) return `${Math.round(ms)}ms`; + const s = ms / 1000; + if (s < 60) return `${trimZero(s)}s`; + const m = Math.floor(s / 60); + return `${m}m${Math.round(s % 60)}s`; +} + +export function formatAbort(p: AbortPayload, t: Messages): string { + return dim(t.abortLabel(p.reason ?? undefined)); +} + +/** + * Statically renders resumed history messages (`--resume`: full-message semantics, no + * partial_*, including interrupted messages and their markers). Uses the + * same color scheme as streaming rendering: user input `> `, dim thinking, cyan tool + * calls, dim tool-output gutter; a message whose `stop_reason` isn't completed gets a + * dim marker appended at the end of its line. + */ +export function renderHistory( + messages: OmniMessage[], + out: NodeJS.WritableStream, + t: Messages = defaultMessages(), +): void { + for (const msg of messages) { + if (isEventMessage(msg)) { + const p = msg.payload as { type?: string } & AbortPayload; + if (p.type === "abort") out.write(`${formatAbort(p, t)}\n`); + continue; + } + if (!isModelMessage(msg)) continue; + const p = msg.payload as { + type?: string; + role?: string; + text?: string; + thinking?: string; + name?: string; + arguments?: string; + output?: string; + images?: string[]; + tool_call_id?: string; + stop_reason?: string; + }; + const marker = p.stop_reason && p.stop_reason !== "completed" ? dim(` [${p.stop_reason}]`) : ""; + switch (p.type) { + case "text": + if (p.role === "user") out.write(`\n> ${p.text ?? ""}\n`); + else out.write(`${p.text ?? ""}${marker}\n`); + break; + case "image_url": + out.write(`\n> ${dim("[image]")}\n`); + break; + case "thinking": + out.write(`${dim(p.thinking ?? "")}${marker}\n`); + break; + case "tool_call": { + const preview = + renderPartialToolCall(p.name ?? "", p.arguments ?? "") ?? `${p.name} ${p.arguments}`; + out.write(`${cyan(`[${callTag(p.tool_call_id ?? "")}] ${preview}`)}${marker}\n`); + break; + } + case "tool_call_output": { + const tag = callTag(p.tool_call_id ?? ""); + for (const line of (p.output ?? "").split("\n")) { + out.write(`${DIM}[${tag}] >> ${RESET}${line}\n`); + } + // Attached images aren't rendered by the terminal; print one placeholder line per image. + for (const _ of p.images ?? []) { + out.write(`${DIM}[${tag}] >> [image]${RESET}\n`); + } + break; + } + default: + break; // inline_data / inline_thinking etc.: not shown in static history rendering for now + } + } +} + +/** + * Streaming renderer: writes the OmniMessage stream to the output stream. The display + * text for tool calls is decided locally by `tool-render.ts`; it no longer accepts a + * tool-render callback from core (rendering has moved down into the CLI). + */ +export class StreamRenderer { + private readonly out: NodeJS.WritableStream; + private readonly t: Messages; + + /** Pending render queue: while the screen is held (a streaming segment is in progress / awaiting user input), messages queue up here. */ + private pending: OmniMessage[] = []; + /** The streaming segment currently holding the screen ("llm" or "out:"); null = idle. */ + private holder: string | null = null; + /** Awaiting user input (approval prompt): locks the screen, all messages queue up. */ + private promptActive = false; + /** Key of the call the current interactive prompt belongs to (the tool_call passed to beginUserPrompt); null = unattached. */ + private promptKey: string | null = null; + /** + * Approval results for **other calls** that arrive during an interactive prompt + * (concurrent subagent / auto-approval paths): must not be written straight into the + * middle of an unanswered prompt, so they're deferred and rendered in order once + * endUserPrompt unlocks the screen. + */ + private deferredDecisions: Array<{ + toolCall: OmniMessage; + decision: ApprovalDecision; + }> = []; + /** Reentrancy guard for drain. */ + private draining = false; + /** + * Keys (origin chain + tool_call_id) of call lines already **rendered in place** from + * a complete message: rendered ahead of the streaming copy at approval time, so any + * streaming/nested copy that arrives afterward is deduplicated and skipped based on + * this set. Guarantees the approval prompt always immediately follows its matching + * call line (messages arrive through an async pipeline and may arrive later than the + * approval callback). Cleared at task end (see endTask). + */ + private ensuredCallLines = new Set(); + /** Call-line key of the last **content line actually written**; cleared once anything else is written. Used to check whether a call line is still adjacent to the current position. */ + private lastLineKey: string | null = null; + /** Calls whose result has already been rendered in place at the approval callback (keyed the same as callLineKey); deduplicates a later-arriving approval_decision event. */ + private renderedDecisions = new Set(); + + /** Whether we're currently mid-way through a streaming line (text/thinking/tool output) that hasn't been newline-terminated yet. */ + private inLine = false; + /** Whether we're currently in a dim span (thinking), used to know when to emit RESET. */ + private inDim = false; + /** Whether tool-call output is at the start of a line (decides whether the gutter needs to be written). */ + private toolOutLineStart = true; + /** Buffer for partial_tool_call; each delta streams out the newly appended suffix of the preview. */ + private partialToolCalls = new Map< + string, + { name: string; arguments: string; lastPreview: string } + >(); + /** The partial_tool_call currently being rendered as a stream. */ + private partialToolCallLineId: string | null = null; + /** This task's accumulated request tokens, the parent session's cumulative Session tokens, and whether this task has seen any usage. */ + private taskTokens = 0; + private sessionTotal = 0; + private hasUsage = false; + /** + * Session-level accumulation of sub-session (subagent) request tokens: persists across + * tasks, never reset by endTask. The Token total shown to the user = + * sessionTotal + subagentTotal, using the same accounting as this task's delta + * (parent + child), guaranteeing the sum of per-task deltas never exceeds the + * cumulative increase. + */ + private subagentTotal = 0; + /** Current context (= input+output = total of the most recent request), the context at the end of the previous task, and cumulative Session elapsed time (ms). */ + private contextNow = 0; + private contextAtTaskStart = 0; + private sessionElapsedMs = 0; + /** + * Compaction in progress (between a pair of parent-session compaction events): any + * parent-session token_usage arriving during this window is compaction-request usage — + * it does not update the context accounting (the actual usage after compaction is + * reported by the next normal request); it's accumulated into compactionTokens so the + * compaction-completion line can show "usage this time", and also staged into + * pendingCompactionTokens pending final attribution (see below). + */ + private compactionActive = false; + private compactionTokens = 0; + /** + * Staged compaction usage: when a compaction event arrives, it's not yet known whether + * it happened **mid-turn** (a normal request_end still follows in this turn -> + * attribute to this turn) or **after the turn ended** (nothing follows -> don't + * attribute to this turn). Mid-turn compaction is folded into taskTokens at the next + * non-compaction request_end; compaction after the turn ended is discarded when + * endTask/endCompact settles up. Uses the same accounting as the Web side + * (stream-model / task-stats). + */ + private pendingCompactionTokens = 0; + /** + * Timestamps (ms) of this task's first (non-session_meta) message and its last + * **non-compaction** request_end: the elapsed time shown in the stats line = the + * latter minus the former. A mid-turn compaction naturally falls within this span and + * is counted; one after the turn ends falls after it and is naturally excluded + * (consistent with "the last request_end before stats were queried"). The degenerate + * case of a turn with no request_end at all falls back to the externally supplied + * wall-clock elapsed time. + */ + private taskFirstTsMs: number | null = null; + private taskLastReqEndMs: number | null = null; + /** Terminal state (timeout/malformed) of the previous request: the next request_begin is a retry, at which point a notice is printed. */ + private pendingRetry: "timeout" | "malformed" | null = null; + /** Number of retries already initiated (increments on consecutive failures, reset once a request completes normally). */ + private reconnectRun = 0; + + constructor(out: NodeJS.WritableStream = process.stdout, t: Messages = defaultMessages()) { + this.out = out; + this.t = t; + } + + handle(msg: OmniMessage): void { + this.pending.push(msg); + this.drain(); + } + + /** + * Enters user interaction (approval prompt): first ensures the call line awaiting + * approval is **immediately adjacent to the current position** (if unrendered or + * separated by other output, render it in place from the complete message directly), + * then finishes the current line and locks the screen, queuing any messages that + * arrive in the meantime — guaranteeing "tool call -> approval prompt" stay adjacent, + * for both the main Agent and subagents. + */ + beginUserPrompt(toolCall?: OmniMessage): void { + if (toolCall) this.ensureAdjacentCallLine(toolCall); + this.finishLine(); + this.promptActive = true; + this.promptKey = toolCall + ? this.callLineKey(toolCall.payload.tool_call_id, toolCall.origin) + : null; + } + + /** + * Renders one approval result, guaranteeing "tool call -> (approval prompt) -> + * approval result" appear consecutively: + * - interactive path: called **before** the prompt ends and unlocks (nothing else can + * preempt output while the lock is held); + * - auto-approval path (allow-all etc., no prompt): if the call line isn't adjacent, + * render it in place first, then write the result, so they appear as a pair. + * Idempotent (a given call's result is rendered only once); a subsequent + * approval_decision event arriving through the pipeline is deduplicated by key. + */ + noteApprovalDecision(toolCall: OmniMessage, decision: ApprovalDecision): void { + const key = this.callLineKey(toolCall.payload.tool_call_id, toolCall.origin); + // The screen is locked by **another call's** interactive prompt (e.g. auto-approval + // of a concurrent subagent): must not write straight into the middle of an + // unanswered prompt, so defer until unlocked; this prompt's own result still renders + // in place as usual (it holds the lock). + if (this.promptActive && this.promptKey !== key) { + this.deferredDecisions.push({ toolCall, decision }); + return; + } + if (this.renderedDecisions.has(key)) return; + this.renderedDecisions.add(key); + this.ensureAdjacentCallLine(toolCall); + this.finishLine(); + this.out.write(`${dim(this.t.approvalDecision(decision))}\n`); + this.lastLineKey = null; + } + + /** Call-line dedup key: origin chain + tool_call_id (parent/child session ids may collide, so the chain is needed to disambiguate). */ + private callLineKey(id: string, origin?: readonly MessageOrigin[]): string { + return `${origin?.join("/") ?? ""}:${id}`; + } + + /** + * Ensures a given tool_call's call line is adjacent to the current position: if it + * isn't the last content line (unrendered, or separated by other output since), it is + * (re-)rendered in place from the complete message, and registered so any late + * streaming/nested copy is deduplicated and skipped. + */ + private ensureAdjacentCallLine(tc: OmniMessage): void { + const key = this.callLineKey(tc.payload.tool_call_id, tc.origin); + // The call line is already the last content line and its streaming segment has + // already finished: already adjacent, nothing to do. If it's still mid-stream (the + // line may show only half the arguments), re-render the full line in place and + // register it for dedup — otherwise a late tail delta arriving after unlock would + // start a duplicate call line, breaking the "call -> prompt -> result" adjacency + // invariant. + if (this.lastLineKey === key && this.partialToolCallLineId !== tc.payload.tool_call_id) { + return; + } + this.renderCallLine(tc.payload, tc.origin, key); + } + + /** Renders one call line in place from a complete tool_call and registers its dedup key (shared by in-place approval rendering and nested rendering). */ + private renderCallLine( + p: ToolCallPayload, + origin: readonly MessageOrigin[] | undefined, + key: string, + ): void { + this.ensuredCallLines.add(key); + const preview = renderPartialToolCall(p.name, p.arguments) ?? `${p.name} ${p.arguments}`; + this.finishLine(); + this.out.write(`${cyan(`[${callTag(p.tool_call_id, origin)}] ${preview}`)}\n`); + this.lastLineKey = key; + } + + /** User interaction ends: unlocks the screen, first renders approval results deferred during the lock, then drains the queue. */ + endUserPrompt(): void { + this.promptActive = false; + this.promptKey = null; + this.flushDeferredDecisions(); + this.drain(); + } + + /** Renders approval results deferred during the interactive prompt (call line + result as a pair; called after unlocking). */ + private flushDeferredDecisions(): void { + const deferred = this.deferredDecisions; + if (deferred.length === 0) return; + this.deferredDecisions = []; + for (const d of deferred) this.noteApprovalDecision(d.toolCall, d.decision); + } + + /** Streaming segment ownership: the LLM stream (text/thinking/tool_call share one stream serially) or a given tool's output stream; null = atomic message. */ + private streamOwner(msg: OmniMessage): string | null { + if (msg.origin && msg.origin.length > 0) return null; // nested messages render as atomic lines + if (!isModelMessage(msg)) return null; + const type = msg.payload.type; + if (type === "partial_text" || type === "partial_thinking" || type === "partial_tool_call") { + return "llm"; + } + if (type === "partial_tool_call_output") { + return `out:${(msg.payload as PartialToolCallOutputPayload).tool_call_id}`; + } + return null; + } + + private isStop(msg: OmniMessage): boolean { + return (msg.payload as { event_type?: string }).event_type === "stop"; + } + + /** + * Drains the pending render queue. The same streaming segment (start->delta->stop) + * holds the screen until stop, while other messages queue up; while the screen is + * held, the holder's own subsequent messages are let through first (preserving + * in-segment order, while other messages keep their arrival order); nothing is let + * through while awaiting user input. + */ + private drain(): void { + if (this.draining) return; + this.draining = true; + try { + while (!this.promptActive && this.pending.length > 0) { + if (this.holder === null) { + const msg = this.pending.shift()!; + const owner = this.streamOwner(msg); + if (owner !== null) this.holder = this.isStop(msg) ? null : owner; + this.renderNow(msg); + continue; + } + // Screen is held: let through all of the holder's own messages in a single + // pass (avoiding the quadratic cost of rescanning from the queue head after + // each message); once the holder releases mid-scan (stop), put the remaining + // messages back in original order, returning to plain FIFO. + const keep: OmniMessage[] = []; + let progressed = false; + for (let i = 0; i < this.pending.length; i++) { + if (this.promptActive || this.holder === null) { + keep.push(...this.pending.slice(i)); + break; + } + const msg = this.pending[i]!; + if (this.streamOwner(msg) === this.holder) { + if (this.isStop(msg)) this.holder = null; + this.renderNow(msg); + progressed = true; + } else { + keep.push(msg); + } + } + this.pending = keep; + if (!progressed) break; // no message from the holder in the queue: wait for it to arrive + } + } finally { + this.draining = false; + } + } + + /** Actually renders one message (queue scheduling is already done by drain). */ + private renderNow(msg: OmniMessage): void { + if (msg.origin && msg.origin.length > 0) { + this.handleNested(msg); + return; + } + // The timestamp of this task's first (non-session_meta) message = the start point for + // the stats-line elapsed time. session_meta can predate this turn by a long time (a + // session may sit idle for a day before the first question), so it is excluded, + // matching Web / Trace accounting. + if (this.taskFirstTsMs === null && msg.type !== "session_meta") { + const ms = Date.parse(msg.timestamp); + if (Number.isFinite(ms)) this.taskFirstTsMs = ms; + } + if (isModelMessage(msg)) { + const payload = msg.payload; + switch (payload.type) { + case "partial_text": + this.handlePartialText(payload as PartialTextPayload); + return; + case "partial_thinking": + this.handlePartialThinking(payload as PartialThinkingPayload); + return; + case "partial_tool_call": + this.handlePartialToolCall(payload as PartialToolCallPayload); + return; + case "partial_tool_call_output": + this.handlePartialToolOutput(payload as PartialToolCallOutputPayload); + return; + // Complete (non-streaming) model_msg is never rendered (including image_url/inline_*); the content has already been shown by partial_*. + default: + return; + } + } + + if (isEventMessage(msg)) { + const payload = msg.payload; + if (payload.type === "token_usage") { + // Accumulate this task's usage, printed together when the task ends (endTask), + // not shown after every tool call/round. + const p = payload as TokenUsagePayload; + this.sessionTotal = p.session.total; + if (this.compactionActive) { + // Usage of a compaction request: staged first (final attribution depends on + // whether a normal request_end still follows in this turn), and accumulated + // into compactionTokens so the compaction-completion line can show "usage this + // time"; does not update context accounting (see the compactionActive comment). + this.pendingCompactionTokens += p.request.total; + this.compactionTokens += p.request.total; + } else { + this.taskTokens += p.request.total; + this.contextNow = p.request.total; // current context = total of the most recent normal request + this.hasUsage = true; + } + } else if (payload.type === "approval_decision") { + // The approval result has usually already been rendered in place at the + // approval callback (noteApprovalDecision, guaranteeing three consecutive + // lines); deduplicated here by key; falls back to rendering one line (without a + // pairing tag) if it wasn't rendered yet. + const p = payload as ApprovalDecisionPayload; + if (this.renderedDecisions.delete(this.callLineKey(p.tool_call_id))) return; + this.finishLine(); + this.out.write(`${dim(this.t.approvalDecision(p.decision))}\n`); + this.lastLineKey = null; + } else if (payload.type === "abort") { + // Run ended (user interrupt / retries exhausted): clear any pending retry state so the next run doesn't mistakenly print a retry line. + this.pendingRetry = null; + this.reconnectRun = 0; + this.finishLine(); + this.out.write(`${formatAbort(payload as AbortPayload, this.t)}\n`); + this.lastLineKey = null; + } else if (payload.type === "request_begin") { + // The previous request ended in timeout/malformed -> this request is a retry + // carrying : printed when the retry **actually starts** (when + // retries are exhausted, there's no retry after the last failure, only an abort + // explaining why). + if (this.pendingRetry) { + this.reconnectRun += 1; + this.finishLine(); + this.out.write(`${dim(this.t.reconnectLabel(this.pendingRetry, this.reconnectRun))}\n`); + this.lastLineKey = null; + this.pendingRetry = null; + } + } else if (payload.type === "request_end") { + const p = payload as RequestEndPayload; + if (!this.compactionActive) { + // A non-compaction request_end = the end of the turn so far: records the + // timestamp (the end point for elapsed time), and settles any previously + // staged compaction usage — reaching here means that compaction was followed + // by a normal Request in this turn (mid-turn compaction), so its usage is + // attributed to this turn. + const ms = Date.parse(msg.timestamp); + if (Number.isFinite(ms)) this.taskLastReqEndMs = ms; + if (this.pendingCompactionTokens > 0) { + this.taskTokens += this.pendingCompactionTokens; + this.pendingCompactionTokens = 0; + this.hasUsage = true; + } + } + if (p.status === "timeout" || p.status === "malformed") { + this.pendingRetry = p.status; + } else { + this.pendingRetry = null; + this.reconnectRun = 0; + } + } else if (payload.type === "compaction_begin") { + // Paired compaction events: begin signals compaction is in progress. + const p = payload as CompactionBeginPayload; + this.finishLine(); + this.compactionActive = true; + this.compactionTokens = 0; + this.out.write(`${dim(this.t.compactionStart(p.mode, p.reason))}\n`); + this.lastLineKey = null; + } else if (payload.type === "compaction_end") { + // end signals the result and shows the tokens consumed by the compaction request (if any). + const p = payload as CompactionEndPayload; + this.finishLine(); + this.compactionActive = false; + // Same accounting as the stats line: total = Session cumulative (parent + child), delta = usage of this compaction. + const tokens = + this.compactionTokens > 0 + ? { + total: humanizeTokens(this.sessionTotal + this.subagentTotal), + delta: signedDelta(humanizeTokens(this.compactionTokens)), + } + : undefined; + this.compactionTokens = 0; + this.out.write(`${dim(this.t.compactionStop(p.mode, p.status, tokens))}\n`); + this.lastLineKey = null; + } + return; + } + // session_meta: not rendered. + } + + /** + * Nested sub-session messages (carrying an origin): renders the child tool call + * (tagged `agent-xxx-tool-xxx` to mark it as coming from a subagent) and its approval + * result; the request delta of a child token_usage counts toward this task's usage; + * everything else is not rendered (see the rendering rule at the top of this file). + */ + private handleNested(msg: OmniMessage): void { + const origin = msg.origin!; + if (isModelMessage(msg)) { + if (msg.payload.type === "tool_call") { + // A complete tool_call renders one line (nested messages never render + // partial_*, so there's no duplication); one already rendered in place at + // approval time (message arrived later than the approval callback) is + // deduplicated by key and skipped. + const p = msg.payload as ToolCallPayload; + const key = this.callLineKey(p.tool_call_id, origin); + if (this.ensuredCallLines.has(key)) return; + this.renderCallLine(p, origin, key); + } + return; + } + if (isEventMessage(msg)) { + if (msg.payload.type === "approval_decision") { + // The approval result is usually already rendered in place at the approval callback; deduplicated here by key; falls back to rendering if it wasn't rendered yet. + const p = msg.payload as ApprovalDecisionPayload; + if (this.renderedDecisions.delete(this.callLineKey(p.tool_call_id, origin))) { + return; + } + this.finishLine(); + this.out.write(`${dim(this.t.approvalDecision(p.decision))}\n`); + this.lastLineKey = null; + } else if (msg.payload.type === "token_usage") { + // Child-session usage counts toward this task's Token delta and the Session total (parent and child use the same accounting); context still follows parent-session accounting. + const req = (msg.payload as TokenUsagePayload).request.total; + this.taskTokens += req; + this.subagentTotal += req; + this.hasUsage = true; + } + } + } + + private handlePartialText(p: PartialTextPayload): void { + if (p.event_type === "stop") { + this.finishLine(); + return; + } + // Insert a line break when switching from thinking (dim) to body text, to avoid them running together. + if (this.inDim) this.finishLine(); + if (p.text) { + this.out.write(p.text); + this.inLine = true; + this.lastLineKey = null; + } + } + + private handlePartialThinking(p: PartialThinkingPayload): void { + if (p.event_type === "stop") { + this.finishLine(); + return; + } + if (!this.inDim) { + this.out.write(DIM); + this.inDim = true; + } + if (p.thinking) { + this.out.write(p.thinking); + this.inLine = true; + this.lastLineKey = null; + } + } + + private handlePartialToolCall(p: PartialToolCallPayload): void { + // The call line was already rendered in place from the complete message at approval time: skip the whole late-arriving streaming copy (clean up the buffer on stop). + if (this.ensuredCallLines.has(this.callLineKey(p.tool_call_id))) { + if (p.event_type === "stop") this.partialToolCalls.delete(p.tool_call_id); + return; + } + let partial = this.partialToolCalls.get(p.tool_call_id); + if (!partial) { + if (p.event_type === "stop") return; + partial = { name: p.name, arguments: "", lastPreview: "" }; + this.partialToolCalls.set(p.tool_call_id, partial); + } + if (p.name) partial.name = p.name; + if (p.arguments) { + partial.arguments += p.arguments; + } + + if (p.event_type === "stop") { + if (partial.lastPreview) this.finishLine(); + this.partialToolCalls.delete(p.tool_call_id); + return; + } + + if (!p.arguments) return; + + if (this.inDim) this.finishLine(); + const preview = renderPartialToolCall(partial.name, partial.arguments); + if (preview === null) return; + + // The line starts with a pairing tag [tool-], matching the output line that follows. + const key = this.callLineKey(p.tool_call_id); + if (this.partialToolCallLineId !== p.tool_call_id) { + this.finishLine(); + this.partialToolCallLineId = p.tool_call_id; + this.out.write(cyan(`[${callTag(p.tool_call_id)}] ${preview}`)); + } else if (preview.startsWith(partial.lastPreview)) { + this.out.write(cyan(preview.slice(partial.lastPreview.length))); + } else { + // The preview usually grows monotonically with the arguments; if escaping/folding makes it non-appendable, start a new line with the current readable state. + this.finishLine(); + this.partialToolCallLineId = p.tool_call_id; + this.out.write(cyan(`[${callTag(p.tool_call_id)}] ${preview}`)); + } + partial.lastPreview = preview; + this.inLine = true; + this.lastLineKey = key; + } + + private handlePartialToolOutput(p: PartialToolCallOutputPayload): void { + if (p.event_type === "stop") { + this.finishLine(); + return; + } + if (this.inDim) this.finishLine(); + if (p.output) this.writeToolOutput(p.output, callTag(p.tool_call_id)); + // Image delta (carried whole in a single delta): the terminal doesn't render the + // image itself, so print one placeholder line per image, using the same pairing tag + // as the output gutter. + if (p.images && p.images.length > 0) { + this.finishLine(); + const tag = callTag(p.tool_call_id); + for (const _ of p.images) { + this.out.write(`${DIM}[${tag}] >> [image]${RESET}\n`); + } + this.lastLineKey = null; + } + } + + /** + * Writes tool-call **output** line by line, each line starting with the dim gutter + * `[tool-] >> `, paired with the call line (cyan `[tool-xxx] $ + * cmd`). Streaming chunks arrive incrementally; whether to write the gutter is + * decided by the current line-start state. + */ + private writeToolOutput(chunk: string, tag: string): void { + let i = 0; + while (i < chunk.length) { + if (this.toolOutLineStart) { + this.out.write(`${DIM}[${tag}] >> ${RESET}`); + this.toolOutLineStart = false; + this.inLine = true; + } + const nl = chunk.indexOf("\n", i); + if (nl === -1) { + this.out.write(chunk.slice(i)); + i = chunk.length; + } else { + this.out.write(chunk.slice(i, nl + 1)); + this.toolOutLineStart = true; + this.inLine = false; + i = nl + 1; + } + } + this.lastLineKey = null; + } + + /** + * Task end: forcibly releases the screen lock and drains any remaining messages + * (normally every streaming segment has already closed), finishes the current line, + * and prints one line of stats — all as Session cumulative values + this task's + * delta: context (input+output of the most recent request; delta = minus the context + * at the start of this task, which can be negative once compaction shrinks context), + * Token (Session cumulative = parent-session cumulative + child-session cumulative; + * delta = added this task, same accounting for parent and child), elapsed time + * (Session total elapsed; delta = this task's elapsed). This task's counters are then + * reset. + */ + endTask(elapsedMs = 0): void { + this.promptActive = false; + this.promptKey = null; + this.flushDeferredDecisions(); + this.holder = null; + this.drain(); + this.finishLine(); + // This task's elapsed time = first message -> last non-compaction request_end + // (mid-turn compaction falls within the span and is counted; compaction after the + // turn ends falls after it and isn't). The degenerate case of a turn with no + // request_end at all (e.g. aborted before the first Request even ran) falls back to + // the externally supplied wall-clock elapsedMs. Any staged but unsettled compaction + // usage is discarded here (compaction after the turn ended isn't attributed to it). + const elapsed = + this.taskFirstTsMs !== null && this.taskLastReqEndMs !== null + ? Math.max(0, this.taskLastReqEndMs - this.taskFirstTsMs) + : elapsedMs; + this.sessionElapsedMs += elapsed; + if (this.hasUsage) { + const contextDelta = this.contextNow - this.contextAtTaskStart; + this.out.write( + `${dim( + this.t.taskStats({ + context: humanizeTokens(this.contextNow), + contextDelta: signedDelta(humanizeTokens(contextDelta)), + tokens: humanizeTokens(this.sessionTotal + this.subagentTotal), + tokensDelta: signedDelta(humanizeTokens(this.taskTokens)), + elapsed: humanizeDuration(this.sessionElapsedMs), + elapsedDelta: signedDelta(humanizeDuration(elapsed)), + }), + )}\n`, + ); + this.contextAtTaskStart = this.contextNow; + this.lastLineKey = null; + } + this.taskTokens = 0; + this.pendingCompactionTokens = 0; + this.taskFirstTsMs = null; + this.taskLastReqEndMs = null; + this.hasUsage = false; + // Compaction always closes within run/compact (stop is always reached); this is a + // defensive reset to prevent state from leaking into the next task on an + // exceptional path. + this.compactionActive = false; + this.compactionTokens = 0; + // Dedup/buffer registrations are only meaningful within this task: clear them to prevent unbounded growth in long sessions (chat). + this.ensuredCallLines.clear(); + this.renderedDecisions.clear(); + this.partialToolCalls.clear(); + } + + /** + * Cleans up after a manual `/compact` (outside a Task boundary): compaction usage has + * already been shown on the compaction-completion line and counted into the Session + * total, so no stats line is printed here; only settles the Session elapsed time and + * resets this task's counters — otherwise the compaction's usage would remain in + * taskTokens and be mistakenly counted into the next task's `[stats]` delta (or never + * settled at all if the user exits right after). + */ + endCompact(elapsedMs = 0): void { + this.sessionElapsedMs += elapsedMs; + this.taskTokens = 0; + this.pendingCompactionTokens = 0; + this.taskFirstTsMs = null; + this.taskLastReqEndMs = null; + this.hasUsage = false; + this.compactionActive = false; + this.compactionTokens = 0; + } + + private closeDim(): void { + if (this.inDim) { + this.out.write(RESET); + this.inDim = false; + } + } + + /** Finishes the current streaming line: closes dim mode, emits a trailing newline, and resets tool output to line-start. */ + private finishLine(): void { + this.closeDim(); + if (this.inLine) { + this.out.write("\n"); + this.inLine = false; + } + this.toolOutLineStart = true; + this.partialToolCallLineId = null; + } +} diff --git a/packages/cli/src/task-loop.ts b/packages/cli/src/task-loop.ts new file mode 100644 index 0000000..823ea3f --- /dev/null +++ b/packages/cli/src/task-loop.ts @@ -0,0 +1,107 @@ +/** + * Consumption loop that drives a Task to completion (CLI side, shared by run and chat). + * + * New protocol: `session.run(prompt, { signal, approve })` runs the entire ReAct loop in one + * call — within a turn, the engine invokes the `approve` callback for each tool_call, executing + * it on allow, with execution possibly overlapping. The CLI only needs to consume the output + * stream and supply `approve`. The approval strategy is determined by the permission mode + * (allow-all / deny-all / read-only / always-ask per-call approval). + */ +import { isEventMessage } from "@prismshadow/penguin-core"; +import type { ApproveFn, OmniMessage, Session } from "@prismshadow/penguin-core"; +import type { StreamRenderer } from "./render.js"; +import { makeApprove, promptApproval, type ApprovalMode } from "./approval.js"; +import type { Messages } from "./i18n.js"; + +export interface RunTaskOptions { + /** Approval mode (default allow-all). */ + mode?: ApprovalMode; + /** Interrupt signal (Ctrl-C, etc.). */ + signal?: AbortSignal; + renderer: StreamRenderer; + /** The actual Q&A for interactive approval; defaults to the one-off `promptApproval`. */ + interactivePrompt?: ApproveFn; + /** Message set. */ + t: Messages; +} + +/** Result of one Task: `aborted` = the Task ended with an abort event (LLM failure/reconnect exhausted/user interrupt). */ +export interface RunTaskResult { + aborted: boolean; +} + +export async function runTask( + session: Session, + prompt: OmniMessage[], + opts: RunTaskOptions, +): Promise { + const basePrompt: ApproveFn = opts.interactivePrompt ?? (() => promptApproval({ t: opts.t })); + // Lock the renderer while waiting for the user's approval input: messages from concurrent + // tools/subsessions are queued and released together once the Q&A finishes, so the prompt + // isn't scrambled by later output. The pending tool_call is passed in so its call line stays + // right before the prompt; the approval result is rendered in place **before unlocking** — + // "tool call → approval prompt → approval result" stays three consecutive lines, for both + // the main Agent and subagents (messages arriving via the async pipeline may lag behind the + // approval callback, hence render-in-place plus de-duplication of the copy). + // + // Serialization: the parent session and a run_subagent child session share this callback and + // may request approval concurrently (the parent is waiting on one approval while an + // already-approved child session starts its own). Concurrent prompts would clobber the same + // Q&A state and fight over the same stdin (one answer resolving two questions, leaving the + // other permanently stuck); a promise chain queues them so only one question is asked at a + // time. + let promptChain: Promise = Promise.resolve(); + const interactivePrompt: ApproveFn = (tc) => { + const result = promptChain.then(async () => { + opts.renderer.beginUserPrompt(tc); + try { + const decision = await basePrompt(tc); + opts.renderer.noteApprovalDecision(tc, decision); + return decision; + } finally { + opts.renderer.endUserPrompt(); + } + }); + promptChain = result.then( + () => undefined, + () => undefined, + ); + return result; + }; + const approveByMode = makeApprove({ + mode: opts.mode ?? "allow-all", + toolPermission: (name) => session.toolPermission(name), + interactivePrompt, + }); + // The auto-approval path (allow-all / deny-all / read-only approvals) has no prompt: it + // likewise renders the "call line → approval result" pair in place; the interactive path's + // already-rendered copy is idempotently de-duplicated inside note. + const approve: ApproveFn = async (tc) => { + const decision = await approveByMode(tc); + opts.renderer.noteApprovalDecision(tc, decision); + return decision; + }; + + // A single run drives the whole ReAct loop (the engine requests approval per call and runs + // tools concurrently within a turn). Once the task ends (including on error), endTask + // prints this task's stats (context/Token/elapsed time). The engine collapses failures + // (auth errors, reconnect exhausted, etc.) into a main-session abort event rather than + // throwing; the result reported here reflects that, for `penguin run` to map to + // an exit code. + const startedAt = Date.now(); + let aborted = false; + try { + for await (const msg of session.run(prompt, { + approve, + ...(opts.signal ? { signal: opts.signal } : {}), + })) { + if (isEventMessage(msg) && msg.payload.type === "abort" && (msg.origin?.length ?? 0) === 0) { + aborted = true; + } + opts.renderer.handle(msg); + } + } finally { + opts.renderer.endTask(Date.now() - startedAt); + } + return { aborted }; +} diff --git a/packages/cli/src/tool-render.ts b/packages/cli/src/tool-render.ts new file mode 100644 index 0000000..ad4ca76 --- /dev/null +++ b/packages/cli/src/tool-render.ts @@ -0,0 +1,151 @@ +/** + * Streaming tool-call rendering (CLI side). + * + * The CLI only consumes `partial_tool_call` for visible rendering. exec_command is shown as + * `$ ` as early as possible; input_command / input_subagent show the target session id, + * with a non-empty payload (chars / prompt) appended as `<< ` — the payload is + * critical for approval and later audit (writing to stdin is equivalent to running a command), + * so the session id alone is not enough; run_subagent shows the prompt; other tools fall back + * to `name(args-prefix)`. + * + * The render layer streams by appending to the preview (see render.ts), so the preview format + * must stay append-only: rendering only starts once the target id has fully appeared, the + * payload is only appended at the end, and the preview stops growing once it hits the + * truncation limit. + */ + +/** Max length of the single-line preview for a payload (chars / prompt); truncated with an ellipsis beyond this, after which the preview stops growing. */ +const MAX_PAYLOAD_PREVIEW = 120; + +/** Collapse to a single line: newlines/runs of whitespace become a single space, and leading/trailing whitespace is trimmed. */ +function toSingleLine(text: string): string { + return text.replace(/\s+/g, " ").trim(); +} + +/** + * Turn control characters into a visible, faithful form so stdin writes don't garble the + * screen: `\n`/`\r`/`\t` are shown as escape literals (whether Enter was pressed is important + * information and must not collapse into a space), other C0 control chars and DEL use caret + * notation (U+0003 → `^C`); backslash itself is escaped to avoid ambiguity. + */ +function visualizeControlChars(text: string): string { + return text.replace(/[\\\u0000-\u001f\u007f]/g, (ch) => { + if (ch === "\\") return "\\\\"; + if (ch === "\n") return "\\n"; + if (ch === "\r") return "\\r"; + if (ch === "\t") return "\\t"; + if (ch === "\u007f") return "^?"; + return `^${String.fromCharCode(ch.charCodeAt(0) + 64)}`; + }); +} + +/** Truncate to the single-line preview limit, appending an ellipsis if exceeded. */ +function capPreview(text: string): string { + return text.length > MAX_PAYLOAD_PREVIEW ? `${text.slice(0, MAX_PAYLOAD_PREVIEW)}…` : text; +} + +/** Extract the current value of a string field from a possibly-incomplete JSON object string. */ +function extractPartialStringField(argsJson: string, field: string): string | null { + const key = `"${field}"`; + const keyIndex = argsJson.indexOf(key); + if (keyIndex === -1) return null; + + let i = keyIndex + key.length; + while (/\s/.test(argsJson[i] ?? "")) i += 1; + if (argsJson[i] !== ":") return null; + i += 1; + while (/\s/.test(argsJson[i] ?? "")) i += 1; + if (argsJson[i] !== '"') return null; + i += 1; + + let out = ""; + let escaped = false; + for (; i < argsJson.length; i += 1) { + const ch = argsJson[i]!; + if (escaped) { + switch (ch) { + case "n": + out += "\n"; + break; + case "r": + out += "\r"; + break; + case "t": + out += "\t"; + break; + case "b": + out += "\b"; + break; + case "f": + out += "\f"; + break; + case '"': + case "\\": + case "/": + out += ch; + break; + case "u": { + // If \uXXXX is cut off at an incremental chunk boundary, return "as far as we got": + // emitting the incomplete hex as a literal would cause a rollback once the next + // increment completes it (breaking append-only preview); the render layer falls + // back to a new line in that case. + if (i + 5 > argsJson.length) return out; + const hex = argsJson.slice(i + 1, i + 5); + if (/^[0-9a-fA-F]{4}$/.test(hex)) { + out += String.fromCharCode(Number.parseInt(hex, 16)); + i += 4; + } + break; + } + default: + out += ch; + break; + } + escaped = false; + continue; + } + if (ch === "\\") { + escaped = true; + continue; + } + if (ch === '"') return out; + out += ch; + } + return out; +} + +/** + * Streaming argument preview: exec_command shows `$ ` once cmd can be read; input_command / + * input_subagent show `⌨ → ` once the target id is available, with a non-empty + * chars / prompt appended as `<< ` (an empty payload just means polling, left as-is); + * run_subagent shows `run_subagent << ` once prompt can be read; other tools fall back + * to name(args-prefix). + */ +export function renderPartialToolCall(name: string, argsJson: string): string | null { + if (!argsJson) return null; + if (name === "exec_command") { + const cmd = extractPartialStringField(argsJson, "cmd"); + if (cmd !== null) return `$ ${toSingleLine(cmd)}`; + return null; + } + if (name === "run_subagent") { + const prompt = extractPartialStringField(argsJson, "prompt"); + if (prompt !== null) return `run_subagent << ${capPreview(toSingleLine(prompt))}`; + return null; + } + if (name === "input_command") { + const pid = extractPartialStringField(argsJson, "process_id"); + if (pid === null) return null; + const chars = extractPartialStringField(argsJson, "chars"); + const payload = chars ? ` << ${capPreview(visualizeControlChars(chars))}` : ""; + return `⌨ input_command → ${toSingleLine(pid)}${payload}`; + } + if (name === "input_subagent") { + const sid = extractPartialStringField(argsJson, "subagent_id"); + if (sid === null) return null; + const prompt = extractPartialStringField(argsJson, "prompt"); + const payload = prompt ? ` << ${capPreview(toSingleLine(prompt))}` : ""; + return `⌨ input_subagent → ${toSingleLine(sid)}${payload}`; + } + return `${name || "tool_call"}(${toSingleLine(argsJson)}`; +} diff --git a/packages/cli/test/approval.test.ts b/packages/cli/test/approval.test.ts new file mode 100644 index 0000000..d4a1a35 --- /dev/null +++ b/packages/cli/test/approval.test.ts @@ -0,0 +1,162 @@ +import { describe, expect, it } from "vitest"; +import { Readable, Writable } from "node:stream"; +import { toolCall } from "@prismshadow/penguin-core"; +import type { OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core"; +import { makeApprove, promptApproval, resolveApprovalMode } from "../src/approval.js"; +import { getMessages } from "../src/i18n.js"; + +const t = getMessages("en"); + +/** An in-memory writable stream that collects everything written to output. */ +function collector(): { stream: Writable; text: () => string } { + let buf = ""; + const stream = new Writable({ + write(chunk, _enc, cb) { + buf += chunk.toString(); + cb(); + }, + }); + return { stream, text: () => buf }; +} + +// mock_read_only_tool is only used for approval-mode tests, not a real tool; its permission is read-only ("r"). +const readTool = (): OmniMessage => + toolCall({ name: "mock_read_only_tool", arguments: '{"path":"a"}', toolCallId: "r1" }); +const writeTool = (): OmniMessage => + toolCall({ name: "exec_command", arguments: '{"cmd":"rm x"}', toolCallId: "w1" }); + +const perms: Record = { + mock_read_only_tool: "r", + exec_command: "rw", +}; +const toolPermission = (name: string): "r" | "rw" | undefined => perms[name]; + +describe("promptApproval", () => { + it('returns "allow" when the user types "y"', async () => { + const { stream, text } = collector(); + const decision = await promptApproval({ + input: Readable.from(["y\n"]), + output: stream, + t, + }); + expect(decision).toBe("allow"); + // Output is exactly the approval prompt itself — no input echo, no repeated tool-call rendering. + expect(text()).toBe("? Approve this tool call? [Y/n] "); + }); + + it('returns "allow" on empty input (Enter) — tool approval defaults to yes', async () => { + const { stream } = collector(); + const decision = await promptApproval({ + input: Readable.from(["\n"]), + output: stream, + t, + }); + expect(decision).toBe("allow"); + }); + + it('returns "allow" for "yes" (case-insensitive, trimmed)', async () => { + const { stream } = collector(); + const decision = await promptApproval({ + input: Readable.from([" YES \n"]), + output: stream, + t, + }); + expect(decision).toBe("allow"); + }); + + it('returns "deny" when the user types "n"', async () => { + const { stream } = collector(); + const decision = await promptApproval({ + input: Readable.from(["n\n"]), + output: stream, + t, + }); + expect(decision).toBe("deny"); + }); + + it('returns "allow" for unrelated input (tool approval defaults to yes)', async () => { + const { stream } = collector(); + const decision = await promptApproval({ + input: Readable.from(["maybe\n"]), + output: stream, + t, + }); + expect(decision).toBe("allow"); + }); + + it('returns "deny" when the input stream ends (EOF) instead of hanging', async () => { + const { stream } = collector(); + const decision = await promptApproval({ + input: Readable.from([]), + output: stream, + t, + }); + expect(decision).toBe("deny"); + }); +}); + +describe("resolveApprovalMode", () => { + it("maps --approve values; defaults to allow-all", () => { + expect(resolveApprovalMode("allow-all", t)).toBe("allow-all"); + expect(resolveApprovalMode("read-only", t)).toBe("read-only"); + expect(resolveApprovalMode("deny-all", t)).toBe("deny-all"); + expect(resolveApprovalMode("always-ask", t)).toBe("always-ask"); + expect(resolveApprovalMode("READ-ONLY", t)).toBe("read-only"); + expect(resolveApprovalMode(undefined, t)).toBe("allow-all"); + }); +}); + +describe("makeApprove permission modes", () => { + it("allow-all → allows everything", async () => { + const approve = makeApprove({ + mode: "allow-all", + toolPermission, + interactivePrompt: async () => "deny", + }); + expect(await approve(readTool())).toBe("allow"); + expect(await approve(writeTool())).toBe("allow"); + }); + + it("deny-all → rejects everything", async () => { + const approve = makeApprove({ + mode: "deny-all", + toolPermission, + interactivePrompt: async () => "allow", + }); + expect(await approve(readTool())).toBe("deny"); + expect(await approve(writeTool())).toBe("deny"); + }); + + it("read-only → auto-allows read-only tools, prompts for the rest", async () => { + let prompted = 0; + const approve = makeApprove({ + mode: "read-only", + toolPermission, + interactivePrompt: async () => { + prompted += 1; + return "deny"; + }, + }); + // Read-only tools are auto-allowed without prompting. + expect(await approve(readTool())).toBe("allow"); + expect(prompted).toBe(0); + // Read-write tools are handed off to the interactive prompt (denied here). + expect(await approve(writeTool())).toBe("deny"); + expect(prompted).toBe(1); + }); + + it("always-ask → always delegates to the interactive prompt", async () => { + let prompted = 0; + const approve = makeApprove({ + mode: "always-ask", + toolPermission, + interactivePrompt: async () => { + prompted += 1; + return "allow"; + }, + }); + expect(await approve(readTool())).toBe("allow"); + expect(await approve(writeTool())).toBe("allow"); + expect(prompted).toBe(2); + }); +}); diff --git a/packages/cli/test/chat.test.ts b/packages/cli/test/chat.test.ts new file mode 100644 index 0000000..8091550 --- /dev/null +++ b/packages/cli/test/chat.test.ts @@ -0,0 +1,45 @@ +import { describe, expect, it } from "vitest"; +import { decideSigint } from "../src/commands/chat.js"; +import { parseApprovalAnswer } from "../src/approval.js"; + +describe("decideSigint (Ctrl-C 行为状态机)", () => { + it("approving → deny(无论缓冲区是否有内容)", () => { + expect(decideSigint("approving", false)).toBe("deny"); + expect(decideSigint("approving", true)).toBe("deny"); + }); + + it("running → abort(中断当前 Task,不退出)", () => { + expect(decideSigint("running", false)).toBe("abort"); + expect(decideSigint("running", true)).toBe("abort"); + }); + + it("idle + 有输入 → clear(清空缓冲区)", () => { + expect(decideSigint("idle", true)).toBe("clear"); + }); + + it("idle + 无输入 → confirm-exit(弹出 y/N 退出确认)", () => { + expect(decideSigint("idle", false)).toBe("confirm-exit"); + }); + + it("confirming-exit → exit(确认中再次 Ctrl-C 直接退出)", () => { + expect(decideSigint("confirming-exit", false)).toBe("exit"); + expect(decideSigint("confirming-exit", true)).toBe("exit"); + }); +}); + +describe("parseApprovalAnswer", () => { + it("y / yes(trim、不区分大小写)→ allow;n / no → deny", () => { + expect(parseApprovalAnswer("y")).toBe("allow"); + expect(parseApprovalAnswer(" YES \n")).toBe("allow"); + expect(parseApprovalAnswer("Y")).toBe("allow"); + expect(parseApprovalAnswer("n")).toBe("deny"); + expect(parseApprovalAnswer("NO")).toBe("deny"); + }); + it("空/无关输入用 fallback(缺省 deny;工具审批传 allow)", () => { + expect(parseApprovalAnswer("")).toBe("deny"); // default fallback + expect(parseApprovalAnswer("nope")).toBe("deny"); + expect(parseApprovalAnswer("", "allow")).toBe("allow"); // tool approval defaults to allow + expect(parseApprovalAnswer("nope", "allow")).toBe("allow"); + expect(parseApprovalAnswer("n", "allow")).toBe("deny"); // explicit n still denies + }); +}); diff --git a/packages/cli/test/config-format.test.ts b/packages/cli/test/config-format.test.ts new file mode 100644 index 0000000..25e09b5 --- /dev/null +++ b/packages/cli/test/config-format.test.ts @@ -0,0 +1,68 @@ +/** + * Unit tests for `config model list` rendering: provider and model_id are separate + * columns (stored fields as-is, with the default model marked `*` before the provider + * column; the request column was removed along with concatenated storage); vision falls + * back to the catalog matched by the (provider, model_id) pair; api_key is masked + * inline; fully empty columns are omitted automatically. + */ +import { describe, expect, it } from "vitest"; +import type { ProjectConfig } from "@prismshadow/penguin-core"; +import { formatModelRows } from "../src/commands/config.js"; + +describe("formatModelRows", () => { + const cfg: ProjectConfig = { + default_model: { provider: "anthropic", model_id: "claude-sonnet-4-6" }, + models: [ + { + provider: "anthropic", + model_id: "claude-sonnet-4-6", + context_window: 1000000, + pricing: { unit: "usd_per_mtok", cache_read: 0.3, cache_write: 3.75, output: 15 }, + }, + { + provider: "custom", + model_id: "my-proxy-model", + client_type: "openai", + vision: false, + api_key: "sk-test-abcd-1234", + }, + ], + }; + + it("provider 与 model_id 双列展示;预置模型 vision 经目录成对匹配,默认模型以 * 标记", () => { + const lines = formatModelRows(cfg); + expect(lines).toHaveLength(2); + expect(lines[0]).toMatch(/^\* anthropic\s+claude-sonnet-4-6\s+vision=Y/); + expect(lines[0]).toContain("price=0.3/3.75/15"); + // The request column was removed; no / concatenation appears anymore. + expect(lines[0]).not.toContain("request="); + expect(lines[0]).not.toContain("anthropic/claude-sonnet-4-6"); + }); + + it("自定义模型 vision 按标注(显式 false 记 -);内联 api_key 掩码显示", () => { + const lines = formatModelRows(cfg); + expect(lines[1]).toMatch(/^ {2}custom\s+my-proxy-model\s+vision=-/); + expect(lines[1]).toContain("client_type=openai"); + expect(lines[1]).toContain("api_key=****1234"); + expect(lines[1]).not.toContain("sk-test-abcd-1234"); + }); + + it("同名 model_id 双 provider 并存时各占一行,默认标记只落在成对命中的那行", () => { + const lines = formatModelRows({ + default_model: { provider: "deepseek", model_id: "m1" }, + models: [ + { provider: "deepseek", model_id: "m1" }, + { provider: "siliconflow", model_id: "m1" }, + ], + }); + expect(lines[0]).toMatch(/^\* deepseek\s+m1\s+vision=Y/); + expect(lines[1]).toMatch(/^ {2}siliconflow\s+m1\s+vision=Y/); + }); + + it("无标注按「缺省=支持」记 Y;全空列省略", () => { + const lines = formatModelRows({ + models: [{ provider: "custom", model_id: "m1" }], + }); + expect(lines[0]).toBe(" custom m1 vision=Y api_key=-"); + }); +}); diff --git a/packages/cli/test/config-model.test.ts b/packages/cli/test/config-model.test.ts new file mode 100644 index 0000000..98e0f44 --- /dev/null +++ b/packages/cli/test/config-model.test.ts @@ -0,0 +1,301 @@ +/** + * Integration tests for `penguin config model add|default|vision|list` (run through + * commander's parseAsync for the full command path): --model-id always takes the + * upstream id, paired with --provider to form a (provider, model_id) reference (add's + * --provider defaults to catalog-based inference, falling back to custom when + * inference fails; default / vision require --provider and raise an error when the + * reference isn't found in models — no string concatenation is ever performed); --root + * specifies the data root directory (takes priority over PENGUIN_HOME); persisted to a + * single hidden .project_config.toml (mode 0600, credentials inline, provider and + * model_id as separate columns); list displays provider and model_id as separate + * columns. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { Command } from "commander"; +import { parse as parseToml } from "smol-toml"; +import { DEFAULT_PROJECT_ID, projectConfigPath } from "@prismshadow/penguin-core"; +import { registerConfigCommand } from "../src/commands/config.js"; +import { getMessages } from "../src/i18n.js"; + +let tmpHome: string; +let tmpRoot: string; +let prevHome: string | undefined; + +beforeEach(async () => { + prevHome = process.env.PENGUIN_HOME; + tmpHome = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-home-")); + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-root-")); + process.env.PENGUIN_HOME = tmpHome; +}); + +afterEach(async () => { + if (prevHome === undefined) delete process.env.PENGUIN_HOME; + else process.env.PENGUIN_HOME = prevHome; + await fs.rm(tmpHome, { recursive: true, force: true }); + await fs.rm(tmpRoot, { recursive: true, force: true }); +}); + +interface TomlModelRef { + provider: string; + model_id: string; +} + +/** + * Runs a `penguin config model …` command, capturing stdout / stderr and the exit code + * (without actually exiting the process; under exitOverride, commander usage errors — + * such as a missing required option — are thrown as a CommanderError, which is + * converted to a non-zero exit code). + */ +async function runModel(args: string[]): Promise<{ out: string; err: string; code: number }> { + const program = new Command(); + program.exitOverride(); + registerConfigCommand(program, getMessages("en")); + const out: string[] = []; + const err: string[] = []; + const outSpy = vi.spyOn(process.stdout, "write").mockImplementation((chunk) => { + out.push(String(chunk)); + return true; + }); + const errSpy = vi.spyOn(process.stderr, "write").mockImplementation((chunk) => { + err.push(String(chunk)); + return true; + }); + const prevExitCode = process.exitCode; + process.exitCode = undefined; + try { + await program.parseAsync(["node", "penguin", "config", "model", ...args]); + return { out: out.join(""), err: err.join(""), code: Number(process.exitCode ?? 0) }; + } catch (e) { + const exitCode = (e as { exitCode?: number }).exitCode; + return { out: out.join(""), err: err.join(""), code: exitCode || 1 }; + } finally { + outSpy.mockRestore(); + errSpy.mockRestore(); + process.exitCode = prevExitCode; + } +} + +describe("penguin config model add/list(--root 与 provider / model_id 分列存储)", () => { + it("--root 优先于 PENGUIN_HOME:落盘到指定根目录的隐藏 .project_config.toml(0600)", async () => { + const add = await runModel([ + "add", + "--model-id", + "my-own-model", + "--api-key", + "sk-root-secret-1", + "--root", + tmpRoot, + ]); + expect(add.code).toBe(0); + // Catalog inference fails -> falls back to the custom group (provider is a separate field, never concatenated into the id). + expect(add.out).toContain("Added model (provider=custom, model_id=my-own-model)."); + + const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID); + expect(path.basename(file)).toBe(".project_config.toml"); + expect((await fs.stat(file)).mode & 0o777).toBe(0o600); + const parsed = parseToml(await fs.readFile(file, "utf8")) as { + models: Array>; + }; + const entry = parsed.models.find( + (m) => m.provider === "custom" && m.model_id === "my-own-model", + ); + expect(entry).toBeDefined(); + expect(entry?.api_key).toBe("sk-root-secret-1"); + // Concatenated storage id and request_model_id have been removed. + expect(entry?.request_model_id).toBeUndefined(); + // The root directory pointed to by PENGUIN_HOME is unaffected. + await expect(fs.access(projectConfigPath(tmpHome, DEFAULT_PROJECT_ID))).rejects.toThrow(); + + // list also reads --root: provider and model_id as separate columns + masked api_key (the request column has been removed). + const list = await runModel(["list", "--root", tmpRoot]); + expect(list.code).toBe(0); + const line = list.out.split("\n").find((l) => l.includes("my-own-model")); + expect(line).toMatch(/custom\s+my-own-model/); + expect(line).toContain("api_key=****et-1"); + expect(list.out).not.toContain("request="); + expect(list.out).not.toContain("sk-root-secret-1"); + }); + + it("内置目录推断分组:上游 id 命中目录时条目落该 provider;--set-default 写成对引用", async () => { + const add = await runModel([ + "add", + "--model-id", + "claude-sonnet-4-6", + "--set-default", + "--root", + tmpRoot, + ]); + expect(add.code).toBe(0); + expect(add.out).toContain("Updated model (provider=anthropic, model_id=claude-sonnet-4-6)."); + expect(add.out).toContain("Default model: (provider=anthropic, model_id=claude-sonnet-4-6)"); + + const parsed = parseToml( + await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"), + ) as unknown as { default_model: TomlModelRef; models: Array> }; + expect(parsed.default_model).toEqual({ + provider: "anthropic", + model_id: "claude-sonnet-4-6", + }); + expect( + parsed.models.find((m) => m.provider === "anthropic" && m.model_id === "claude-sonnet-4-6"), + ).toBeDefined(); + }); + + it("--provider 显式指定分组:同名上游 id 与预置条目互不冲突(各自独立条目)", async () => { + const add = await runModel([ + "add", + "--model-id", + "claude-sonnet-4-6", + "--provider", + "myproxy", + "--base-url", + "https://proxy.example/v1", + "--root", + tmpRoot, + ]); + expect(add.code).toBe(0); + expect(add.out).toContain("Added model (provider=myproxy, model_id=claude-sonnet-4-6)."); + + const list = await runModel(["list", "--root", tmpRoot]); + const line = list.out.split("\n").find((l) => l.includes("myproxy")); + expect(line).toMatch(/myproxy\s+claude-sonnet-4-6/); + expect(line).toContain("base_url=https://proxy.example/v1"); + // The pre-existing anthropic entry remains (the (provider, model_id) pair naturally disambiguates). + expect(list.out.split("\n").some((l) => /anthropic\s+claude-sonnet-4-6/.test(l))).toBe(true); + }); + + it("client_type 缺省按分组语义(PRN-021):custom / 自建 / 网关落 openai,一方厂商不落", async () => { + // custom (catalog inference fails) and self-hosted groups (--provider not a catalog value): default to client_type=openai. + await runModel(["add", "--model-id", "my-openai-proxy", "--root", tmpRoot]); + await runModel(["add", "--model-id", "in-house-1", "--provider", "mylab", "--root", tmpRoot]); + // A non-catalog id under a first-party vendor group: client_type is not set (AgentHub auto-routes by upstream id). + await runModel([ + "add", + "--model-id", + "my-fine-tune", + "--provider", + "deepseek", + "--root", + tmpRoot, + ]); + // Gateway group: openai + the gateway's endpoint base URL pre-filled. + await runModel([ + "add", + "--model-id", + "acme/some-model", + "--provider", + "openrouter", + "--root", + tmpRoot, + ]); + // An explicit --client-type is persisted as-is, not overridden by the default rule. + await runModel([ + "add", + "--model-id", + "special-1", + "--provider", + "mylab", + "--client-type", + "verbatim-type", + "--root", + tmpRoot, + ]); + + const parsed = parseToml( + await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"), + ) as { models: Array> }; + const by = (p: string, id: string) => + parsed.models.find((m) => m.provider === p && m.model_id === id)!; + expect(by("custom", "my-openai-proxy").client_type).toBe("openai"); + expect(by("mylab", "in-house-1").client_type).toBe("openai"); + expect(by("deepseek", "my-fine-tune").client_type).toBeUndefined(); + expect(by("openrouter", "acme/some-model").client_type).toBe("openai"); + expect(by("openrouter", "acme/some-model").base_url).toBe("https://openrouter.ai/api/v1"); + expect(by("mylab", "special-1").client_type).toBe("verbatim-type"); + }); + + it("model default 经 --root 指定根目录设置默认模型(--model-id 上游 id + --provider 成对)", async () => { + const set = await runModel([ + "default", + "--model-id", + "deepseek-v4-flash", + "--provider", + "deepseek", + "--root", + tmpRoot, + ]); + expect(set.code).toBe(0); + expect(set.out).toContain( + "Default model set to (provider=deepseek, model_id=deepseek-v4-flash).", + ); + const parsed = parseToml( + await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"), + ) as unknown as { default_model: TomlModelRef }; + expect(parsed.default_model).toEqual({ + provider: "deepseek", + model_id: "deepseek-v4-flash", + }); + }); +}); + +describe("model default/vision:--provider 必填,(provider, model_id) 成对引用", () => { + it("缺 --provider:commander 用法报错,非零退出码", async () => { + const bad = await runModel(["default", "--model-id", "deepseek-v4-flash", "--root", tmpRoot]); + expect(bad.code).not.toBe(0); + expect(bad.err).toContain("--provider"); + }); + + it("引用落空:成对引用不在 models 中,报错带成对引用与 model list 提示", async () => { + const bad = await runModel([ + "default", + "--model-id", + "no-such-model", + "--provider", + "custom", + "--root", + tmpRoot, + ]); + expect(bad.code).toBe(1); + expect(bad.err).toContain("(provider=custom, model_id=no-such-model)"); + expect(bad.err).toContain("penguin config model list"); + // The upstream id matches a pre-existing entry but --provider names the wrong group: also not found (exact pair, no fuzzy matching). + const wrongGroup = await runModel([ + "vision", + "--model-id", + "claude-sonnet-4-6", + "--provider", + "openai", + "--root", + tmpRoot, + ]); + expect(wrongGroup.code).toBe(1); + expect(wrongGroup.err).toContain("(provider=openai, model_id=claude-sonnet-4-6)"); + expect(wrongGroup.err).toContain("penguin config model list"); + }); + + it("model vision 成对引用命中:设置视觉模型(落盘内联表)", async () => { + const ok = await runModel([ + "vision", + "--model-id", + "claude-sonnet-4-6", + "--provider", + "anthropic", + "--root", + tmpRoot, + ]); + expect(ok.code).toBe(0); + expect(ok.out).toContain( + "Vision model set to (provider=anthropic, model_id=claude-sonnet-4-6).", + ); + const parsed = parseToml( + await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"), + ) as unknown as { vision_model: TomlModelRef }; + expect(parsed.vision_model).toEqual({ + provider: "anthropic", + model_id: "claude-sonnet-4-6", + }); + }); +}); diff --git a/packages/cli/test/config-vault.test.ts b/packages/cli/test/config-vault.test.ts new file mode 100644 index 0000000..9012ff7 --- /dev/null +++ b/packages/cli/test/config-vault.test.ts @@ -0,0 +1,141 @@ +/** + * Integration tests for `penguin config vault set|list|remove` (run through commander's + * parseAsync for the full command path, with PENGUIN_HOME pointed at a temp directory): + * writes to a hidden .vault.toml (mode 0600), list masks values without leaking + * plaintext, remove raises an error on a missing key, --agent-id targets a specific + * Agent, and an invalid key name / an overlong value exit with a non-zero code. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { Command } from "commander"; +import { agentVaultPath, DEFAULT_PROJECT_ID } from "@prismshadow/penguin-core"; +import { registerConfigCommand } from "../src/commands/config.js"; +import { getMessages } from "../src/i18n.js"; + +let tmpRoot: string; +let prevHome: string | undefined; + +beforeEach(async () => { + prevHome = process.env.PENGUIN_HOME; + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-vault-")); + process.env.PENGUIN_HOME = tmpRoot; +}); + +afterEach(async () => { + if (prevHome === undefined) delete process.env.PENGUIN_HOME; + else process.env.PENGUIN_HOME = prevHome; + await fs.rm(tmpRoot, { recursive: true, force: true }); +}); + +/** Runs a `penguin config vault …` command, capturing stdout/stderr and the exit code (without actually exiting the process). */ +async function runVault(args: string[]): Promise<{ out: string; err: string; code: number }> { + const program = new Command(); + program.exitOverride(); + registerConfigCommand(program, getMessages("en")); + const out: string[] = []; + const err: string[] = []; + const outSpy = vi.spyOn(process.stdout, "write").mockImplementation((chunk) => { + out.push(String(chunk)); + return true; + }); + const errSpy = vi.spyOn(process.stderr, "write").mockImplementation((chunk) => { + err.push(String(chunk)); + return true; + }); + const prevExitCode = process.exitCode; + process.exitCode = undefined; + try { + await program.parseAsync(["node", "penguin", "config", "vault", ...args]); + return { out: out.join(""), err: err.join(""), code: Number(process.exitCode ?? 0) }; + } finally { + outSpy.mockRestore(); + errSpy.mockRestore(); + process.exitCode = prevExitCode; + } +} + +describe("penguin config vault", () => { + it("set → list(掩码)→ remove 全链路;落盘为隐藏 .vault.toml 且 0600", async () => { + const set = await runVault(["set", "--key", "MY_KEY", "--value", "vault-secret-9876"]); + expect(set.code).toBe(0); + expect(set.out).toContain("Saved vault entry MY_KEY."); + + const file = agentVaultPath(tmpRoot, DEFAULT_PROJECT_ID, "default_agent"); + expect(path.basename(file)).toBe(".vault.toml"); + expect((await fs.stat(file)).mode & 0o777).toBe(0o600); + expect(await fs.readFile(file, "utf8")).toContain("vault-secret-9876"); + + const list = await runVault(["list"]); + expect(list.code).toBe(0); + expect(list.out).toContain("MY_KEY"); + expect(list.out).toContain("****9876"); + // Plaintext never appears in list output. + expect(list.out).not.toContain("vault-secret-9876"); + + const removed = await runVault(["remove", "--key", "MY_KEY"]); + expect(removed.code).toBe(0); + expect(removed.out).toContain("Removed vault entry MY_KEY."); + const empty = await runVault(["list"]); + expect(empty.out).toContain("The vault is empty."); + }); + + it("--agent-id 定向到目标 Agent 的 vault,不影响 default_agent", async () => { + const set = await runVault([ + "set", + "--key", + "ONLY_A", + "--value", + "va-secret-value-1", + "--agent-id", + "agent-a", + ]); + expect(set.code).toBe(0); + expect( + await fs.readFile(agentVaultPath(tmpRoot, DEFAULT_PROJECT_ID, "agent-a"), "utf8"), + ).toContain("ONLY_A"); + const defaultList = await runVault(["list"]); + expect(defaultList.out).toContain("The vault is empty."); + }); + + it("--root 指定数据根目录(优先于 PENGUIN_HOME),set/list 均定向到该根目录", async () => { + const otherRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-vault-root-")); + try { + const set = await runVault([ + "set", + "--key", + "ROOTED_KEY", + "--value", + "root-secret-value-1", + "--root", + otherRoot, + ]); + expect(set.code).toBe(0); + expect( + await fs.readFile(agentVaultPath(otherRoot, DEFAULT_PROJECT_ID, "default_agent"), "utf8"), + ).toContain("ROOTED_KEY"); + // The root directory pointed to by PENGUIN_HOME is unaffected. + const defaultList = await runVault(["list"]); + expect(defaultList.out).toContain("The vault is empty."); + const rootedList = await runVault(["list", "--root", otherRoot]); + expect(rootedList.out).toContain("ROOTED_KEY"); + } finally { + await fs.rm(otherRoot, { recursive: true, force: true }); + } + }); + + it("非法键名 / 超长值以非零码退出并打印原因;remove 不存在的键报错", async () => { + const badKey = await runVault(["set", "--key", "1BAD", "--value", "v"]); + expect(badKey.code).toBe(1); + expect(badKey.err).toContain("Invalid vault key"); + + const tooLong = await runVault(["set", "--key", "OK_BIG", "--value", "x".repeat(8193)]); + expect(tooLong.code).toBe(1); + expect(tooLong.err).toContain("too long"); + + const ghost = await runVault(["remove", "--key", "GHOST"]); + expect(ghost.code).toBe(1); + expect(ghost.err).toContain("Vault entry GHOST does not exist."); + }); +}); diff --git a/packages/cli/test/i18n.test.ts b/packages/cli/test/i18n.test.ts new file mode 100644 index 0000000..d90f6ce --- /dev/null +++ b/packages/cli/test/i18n.test.ts @@ -0,0 +1,69 @@ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { getMessages, maskApiKey, resolveLanguage } from "../src/i18n.js"; + +describe("resolveLanguage (env PENGUIN_LANG, default en)", () => { + let prev: string | undefined; + beforeEach(() => { + prev = process.env.PENGUIN_LANG; + }); + afterEach(() => { + if (prev === undefined) delete process.env.PENGUIN_LANG; + else process.env.PENGUIN_LANG = prev; + }); + + it("defaults to en when unset", () => { + delete process.env.PENGUIN_LANG; + expect(resolveLanguage()).toBe("en"); + }); + it("matches zh exactly (case-insensitive, trimmed)", () => { + process.env.PENGUIN_LANG = "zh"; + expect(resolveLanguage()).toBe("zh"); + process.env.PENGUIN_LANG = " ZH "; + expect(resolveLanguage()).toBe("zh"); + }); + it("falls back to en for non-exact zh prefixes and anything else", () => { + process.env.PENGUIN_LANG = "zh-CN"; // no longer prefix-matched -> en + expect(resolveLanguage()).toBe("en"); + process.env.PENGUIN_LANG = "fr"; + expect(resolveLanguage()).toBe("en"); + process.env.PENGUIN_LANG = "en"; + expect(resolveLanguage()).toBe("en"); + }); +}); + +describe("getMessages", () => { + it("provides zh and en runtime + help strings", () => { + expect(getMessages("zh").modelAdded("m", "m")).toContain("已添加"); + expect(getMessages("en").modelAdded("m", "m")).toContain("Added"); + expect(getMessages("zh").modelUpdated("m", "m")).toContain("已更新"); + expect(getMessages("en").modelUpdated("m", "m")).toContain("Updated"); + // Command/option descriptions are also localized. + expect(getMessages("zh").config.addDesc).toContain("模型"); + expect(getMessages("en").config.addDesc).toContain("model"); + expect(getMessages("en").run.desc).toContain("Task"); + // config lang copy. + expect(getMessages("zh").config.langDesc).toContain("语言"); + expect(getMessages("en").config.langDesc).toContain("language"); + expect(getMessages("en").langSet("zh", "/x/.zshrc")).toContain("/x/.zshrc"); + expect(getMessages("zh").langInvalid("fr")).toContain("fr"); + }); + + it("header order is agent → workspace → model", () => { + const h = getMessages("en").header("run", "ag", "/ws", "mod"); + expect(h.indexOf("agent=ag")).toBeLessThan(h.indexOf("workspace=/ws")); + expect(h.indexOf("workspace=/ws")).toBeLessThan(h.indexOf("model=mod")); + }); +}); + +describe("maskApiKey", () => { + it("masks all but the last 4 chars", () => { + expect(maskApiKey("sk-1234567890")).toBe("****7890"); + }); + it("fully masks short keys (≤12 chars would leak most of the secret)", () => { + expect(maskApiKey("sk-test-1234")).toBe("***"); + expect(maskApiKey("short")).toBe("***"); + }); + it("returns - when absent", () => { + expect(maskApiKey(undefined)).toBe("-"); + }); +}); diff --git a/packages/cli/test/input.test.ts b/packages/cli/test/input.test.ts new file mode 100644 index 0000000..a5c056d --- /dev/null +++ b/packages/cli/test/input.test.ts @@ -0,0 +1,110 @@ +import { describe, expect, it } from "vitest"; +import { + LineComposer, + PasteFilter, + endsWithContinuation, + splitTrailingPartial, +} from "../src/input.js"; + +/** Feeds a series of input chunks into PasteFilter, collecting the forwarded output and paste events. */ +async function runFilter(chunks: string[]): Promise<{ forwarded: string; pastes: string[] }> { + const filter = new PasteFilter(); + const pastes: string[] = []; + let forwarded = ""; + filter.on("data", (d: Buffer) => { + forwarded += d.toString("utf8"); + }); + filter.on("paste", (t: string) => pastes.push(t)); + for (const c of chunks) filter.write(c); + await new Promise((resolve) => { + filter.end(() => resolve()); + }); + return { forwarded, pastes }; +} + +describe("splitTrailingPartial", () => { + it("holds a trailing partial-marker prefix", () => { + expect(splitTrailingPartial("abc\x1b[200", "\x1b[200~")).toEqual({ + emit: "abc", + hold: "\x1b[200", + }); + }); + it("holds nothing when no trailing prefix", () => { + expect(splitTrailingPartial("hello", "\x1b[200~")).toEqual({ + emit: "hello", + hold: "", + }); + }); +}); + +describe("PasteFilter", () => { + it("forwards normal bytes unchanged", async () => { + const { forwarded, pastes } = await runFilter(["hello\r"]); + expect(forwarded).toBe("hello\r"); + expect(pastes).toEqual([]); + }); + + it("strips markers and emits the pasted block (incl. newlines) as one event", async () => { + const { forwarded, pastes } = await runFilter(["\x1b[200~line1\nline2\nline3\x1b[201~"]); + expect(pastes).toEqual(["line1\nline2\nline3"]); + expect(forwarded).toBe(""); // pasted content is not forwarded to readline + }); + + it("keeps surrounding typed bytes and paste together in order", async () => { + const { forwarded, pastes } = await runFilter(["ab\x1b[200~PASTED\x1b[201~cd\r"]); + expect(forwarded).toBe("abcd\r"); + expect(pastes).toEqual(["PASTED"]); + }); + + it("handles a marker split across chunks", async () => { + const { forwarded, pastes } = await runFilter(["x\x1b[20", "0~mid\x1b[201", "~y\r"]); + expect(forwarded).toBe("xy\r"); + expect(pastes).toEqual(["mid"]); + }); +}); + +describe("endsWithContinuation", () => { + it("odd trailing backslashes → continuation", () => { + expect(endsWithContinuation("foo\\")).toBe(true); + expect(endsWithContinuation("foo\\\\\\")).toBe(true); + }); + it("even/none → not continuation", () => { + expect(endsWithContinuation("foo")).toBe(false); + expect(endsWithContinuation("foo\\\\")).toBe(false); + }); +}); + +describe("LineComposer", () => { + it("single line → immediate message", () => { + const c = new LineComposer(); + expect(c.pushTypedLine("hello")).toEqual({ message: "hello" }); + }); + + it("backslash continuation joins lines with \\n", () => { + const c = new LineComposer(); + expect(c.pushTypedLine("a\\")).toEqual({}); + expect(c.pushTypedLine("b\\")).toEqual({}); + expect(c.pushTypedLine("c")).toEqual({ message: "a\nb\nc" }); + }); + + it("paste buffers a block, Enter on empty line sends it", () => { + const c = new LineComposer(); + expect(c.pushPaste("l1\nl2\n")).toEqual({ lineCount: 2, normalized: "l1\nl2" }); + expect(c.hasPending()).toBe(true); + expect(c.pushTypedLine("")).toEqual({ message: "l1\nl2" }); + expect(c.hasPending()).toBe(false); + }); + + it("paste then typed text appends the text before sending", () => { + const c = new LineComposer(); + c.pushPaste("l1\nl2"); + expect(c.pushTypedLine("more")).toEqual({ message: "l1\nl2\nmore" }); + }); + + it("reset clears pending", () => { + const c = new LineComposer(); + c.pushPaste("a\nb"); + c.reset(); + expect(c.hasPending()).toBe(false); + }); +}); diff --git a/packages/cli/test/lang-config.test.ts b/packages/cli/test/lang-config.test.ts new file mode 100644 index 0000000..71f2a4d --- /dev/null +++ b/packages/cli/test/lang-config.test.ts @@ -0,0 +1,90 @@ +import { mkdtemp, readFile, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, describe, expect, it } from "vitest"; +import { applyLanguageToRc, resolveShellRc, upsertBlock } from "../src/lang-config.js"; + +describe("resolveShellRc", () => { + it("maps zsh / bash / fish to their startup files and syntax", () => { + const zsh = resolveShellRc("/bin/zsh", "/home/u"); + expect(zsh.kind).toBe("zsh"); + expect(zsh.rcPath).toBe("/home/u/.zshrc"); + expect(zsh.body("zh")).toBe("export PENGUIN_LANG=zh"); + + const bash = resolveShellRc("/usr/bin/bash", "/home/u"); + expect(bash.kind).toBe("bash"); + expect(bash.rcPath).toBe("/home/u/.bashrc"); + + const fish = resolveShellRc("/usr/local/bin/fish", "/home/u"); + expect(fish.kind).toBe("fish"); + expect(fish.rcPath).toBe("/home/u/.config/fish/config.fish"); + expect(fish.body("en")).toBe("set -gx PENGUIN_LANG en"); + }); + + it("falls back to ~/.profile for an unknown shell", () => { + const rc = resolveShellRc(undefined, "/home/u"); + expect(rc.kind).toBe("unknown"); + expect(rc.rcPath).toBe("/home/u/.profile"); + }); +}); + +describe("upsertBlock", () => { + it("appends a marked block when none exists", () => { + const out = upsertBlock("export PATH=/x\n", "export PENGUIN_LANG=zh"); + expect(out).toContain("export PATH=/x"); + expect(out).toContain("# >>> PenguinHarness PENGUIN_LANG >>>"); + expect(out).toContain("export PENGUIN_LANG=zh"); + expect(out).toContain("# <<< PenguinHarness PENGUIN_LANG <<<"); + }); + + it("replaces the block in place and is idempotent", () => { + const first = upsertBlock("", "export PENGUIN_LANG=zh"); + const second = upsertBlock(first, "export PENGUIN_LANG=en"); + // Only one block remains, with its content replaced by the latest value. + expect(second.match(/PenguinHarness PENGUIN_LANG/g)?.length).toBe(2); // begin + end markers + expect(second).toContain("export PENGUIN_LANG=en"); + expect(second).not.toContain("export PENGUIN_LANG=zh"); + // Writing the same value again is stable (the block does not keep growing). + const third = upsertBlock(second, "export PENGUIN_LANG=en"); + expect(third).toBe(second); + }); + + it("preserves surrounding content when replacing", () => { + const base = "line1\n" + upsertBlock("", "export PENGUIN_LANG=zh") + "line2\n"; + const out = upsertBlock(base, "export PENGUIN_LANG=en"); + expect(out.startsWith("line1\n")).toBe(true); + expect(out.endsWith("line2\n")).toBe(true); + expect(out).toContain("export PENGUIN_LANG=en"); + }); +}); + +describe("applyLanguageToRc", () => { + let home: string; + afterEach(async () => { + await rm(home, { recursive: true, force: true }); + }); + + it("writes the export line to the resolved startup file", async () => { + home = await mkdtemp(join(tmpdir(), "penguin-lang-")); + const { rcPath, kind } = await applyLanguageToRc("zh", { shell: "/bin/zsh", home }); + expect(kind).toBe("zsh"); + expect(rcPath).toBe(join(home, ".zshrc")); + const content = await readFile(rcPath, "utf8"); + expect(content).toContain("export PENGUIN_LANG=zh"); + + // Switching the language again updates the file in place instead of appending. + await applyLanguageToRc("en", { shell: "/bin/zsh", home }); + const updated = await readFile(rcPath, "utf8"); + expect(updated).toContain("export PENGUIN_LANG=en"); + expect(updated).not.toContain("export PENGUIN_LANG=zh"); + expect(updated.match(/# >>> PenguinHarness/g)?.length).toBe(1); + }); + + it("creates nested config dir for fish", async () => { + home = await mkdtemp(join(tmpdir(), "penguin-lang-")); + const { rcPath } = await applyLanguageToRc("en", { shell: "/usr/bin/fish", home }); + expect(rcPath).toBe(join(home, ".config", "fish", "config.fish")); + const content = await readFile(rcPath, "utf8"); + expect(content).toContain("set -gx PENGUIN_LANG en"); + }); +}); diff --git a/packages/cli/test/render.test.ts b/packages/cli/test/render.test.ts new file mode 100644 index 0000000..219af55 --- /dev/null +++ b/packages/cli/test/render.test.ts @@ -0,0 +1,716 @@ +import { describe, expect, it } from "vitest"; +import { Writable } from "node:stream"; +import { + approvalDecision, + abortEvent, + assistantText, + compactionBegin, + compactionEnd, + requestBegin, + requestEnd, + thinkingMessage, + toolCall, + toolCallOutput, + tokenUsage, + sessionMeta, + partialText, + partialThinking, + partialToolCall, + partialToolCallOutput, + withOrigin, +} from "@prismshadow/penguin-core"; +import type { MessageOrigin } from "@prismshadow/penguin-core"; +import { StreamRenderer, formatAbort, humanizeTokens, renderHistory } from "../src/render.js"; +import { getMessages } from "../src/i18n.js"; + +const t = getMessages("en"); + +function collector(): { stream: Writable; text: () => string } { + let buf = ""; + const stream = new Writable({ + write(chunk, _enc, cb) { + buf += chunk.toString(); + cb(); + }, + }); + return { stream, text: () => buf }; +} + +function stripAnsi(s: string): string { + // eslint-disable-next-line no-control-regex + return s.replace(/\x1b\[[0-9;]*[A-Za-z]/g, ""); +} + +/** Overrides a message's timestamp (the constructor defaults to the current time). */ +function at(ts: string, msg: M): M { + return { ...msg, timestamp: ts }; +} + +/** token_usage shorthand: request.total = req, session.total = sess (all buckets zero, sufficient for this test group). */ +function usage(req: number, sess: number) { + return tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: sess }, + { cache_read: 0, cache_write: 0, output: 0, total: req }, + ); +} + +describe("humanizeTokens", () => { + it("abbreviates with k / M and trims .0", () => { + expect(humanizeTokens(0)).toBe("0"); + expect(humanizeTokens(999)).toBe("999"); + expect(humanizeTokens(1000)).toBe("1k"); + expect(humanizeTokens(1234)).toBe("1.2k"); + expect(humanizeTokens(32000)).toBe("32k"); + expect(humanizeTokens(1_500_000)).toBe("1.5M"); + }); +}); + +describe("pure formatters", () => { + it("formatAbort includes the reason", () => { + expect(stripAnsi(formatAbort({ type: "abort", reason: "ctrl-c" }, t))).toContain("ctrl-c"); + }); + + it("renderHistory includes abort events from resumed sessions", () => { + const { stream, text } = collector(); + renderHistory([assistantText("partial", "aborted"), abortEvent("aborted by user")], stream, t); + expect(stripAnsi(text())).toBe("partial [aborted]\n[abort]: aborted by user\n"); + }); +}); + +describe("StreamRenderer", () => { + it("streams partial_text deltas and does NOT re-render the complete text", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(partialText("start", "Hel")); + r.handle(partialText("delta", "lo ")); + r.handle(partialText("delta", "world")); + r.handle(partialText("stop", "", "completed")); + r.handle(assistantText("Hello world")); // complete message: must not be re-rendered + expect(stripAnsi(text())).toBe("Hello world\n"); + }); + + it("streams partial_thinking (dim) and skips the complete thinking", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(partialThinking("start", "think")); + r.handle(partialThinking("delta", "ing")); + r.handle(partialThinking("stop")); + r.handle(thinkingMessage("thinking")); // must not be re-rendered + expect(stripAnsi(text())).toBe("thinking\n"); + }); + + it("does not render a complete tool_call without partials", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c2" })); + expect(text()).toBe(""); + }); + + it("streams partial_tool_call with a pairing tag and skips the complete tool_call", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "c4" })); + r.handle( + partialToolCall({ eventType: "delta", name: "", arguments: '{"cmd":"l', toolCallId: "c4" }), + ); + r.handle(partialToolCall({ eventType: "delta", name: "", arguments: 's"}', toolCallId: "c4" })); + r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "c4" })); + r.handle(toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c4" })); + // The call line carries a [tool-] pairing tag matching the output line. + expect(stripAnsi(text())).toBe("[tool-c4] $ ls\n"); + }); + + it("streams partial_tool_call_output with a tagged gutter and skips the complete tool_call_output", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(partialToolCallOutput({ eventType: "start", toolCallId: "c3" })); + r.handle(partialToolCallOutput({ eventType: "delta", output: "line1\n", toolCallId: "c3" })); + r.handle(partialToolCallOutput({ eventType: "delta", output: "line2", toolCallId: "c3" })); + r.handle(partialToolCallOutput({ eventType: "stop", toolCallId: "c3" })); + r.handle(toolCallOutput({ output: "line1\nline2", toolCallId: "c3" })); // must not be re-rendered + // Each line starts with a tagged gutter (no indent) matching the call line. + expect(stripAnsi(text())).toBe("[tool-c3] >> line1\n[tool-c3] >> line2\n"); + }); + + it("prints the retry line only when the retry request actually begins", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(requestBegin()); + r.handle(requestEnd("malformed")); + expect(stripAnsi(text())).toBe(""); // the failure itself prints nothing; only the retry's start does + r.handle(requestBegin()); // retry #1 begins + expect(stripAnsi(text())).toContain("retry #1"); + r.handle(requestEnd("timeout")); + r.handle(requestBegin()); // retry #2 begins + expect(stripAnsi(text())).toContain("retry #2"); + // Retry #2 fails again and retries are exhausted: no next request_begin, only abort — no retry #3 appears. + r.handle(requestEnd("malformed")); + r.handle(abortEvent("malformed response failed after 2 retries")); + expect(stripAnsi(text())).not.toContain("retry #3"); + // The first request of the next run is not a retry, so it prints nothing; a new failure after it counts from 1 again. + r.handle(requestBegin()); + r.handle(requestEnd("timeout")); + r.handle(requestBegin()); + const lines = stripAnsi(text()); + expect(lines.match(/retry #1/g)).toHaveLength(2); + expect(lines).not.toContain("retry #3"); + }); + + it("locks the screen to one streaming tool output; other messages queue until its stop", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(partialToolCallOutput({ eventType: "start", toolCallId: "tA" })); + r.handle(partialToolCallOutput({ eventType: "delta", output: "a1\n", toolCallId: "tA" })); + // The screen is locked by tA: other streaming messages queue up. + r.handle(partialText("start", "")); + r.handle(partialText("delta", "hello")); + r.handle(partialToolCallOutput({ eventType: "delta", output: "a2\n", toolCallId: "tA" })); + expect(stripAnsi(text())).toBe("[tool-tA] >> a1\n[tool-tA] >> a2\n"); // hello is still queued + r.handle(partialToolCallOutput({ eventType: "stop", toolCallId: "tA" })); + r.handle(partialText("stop", "", "completed")); + expect(stripAnsi(text())).toBe("[tool-tA] >> a1\n[tool-tA] >> a2\nhello\n"); + }); + + it("queues everything while a user prompt is active and flushes after it ends", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.beginUserPrompt(); + r.handle(partialText("start", "")); + r.handle(partialText("delta", "after prompt")); + r.handle(partialText("stop", "", "completed")); + expect(text()).toBe(""); // the screen is locked while waiting for user input + r.endUserPrompt(); + expect(stripAnsi(text())).toBe("after prompt\n"); + }); + + it("does not print token_usage per turn; endTask prints [stats] line with per-task deltas", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle( + sessionMeta({ + session_id: "s", + provider: "custom", + model_id: "m", + model_context_window: 1, + system_prompt: "sp", + tools: [{ name: "exec_command", description: "test tool" }], + thinking_level: "medium", + agent_state: "/a", + workspace: "/w", + }), + ); + // Two turns: request total 1500, 4000. Per-task token delta = 5500; session cumulative = 12000; + // context = the latest request's input+output (= total) = 4000. + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 8000 }, + { cache_read: 0, cache_write: 0, output: 200, total: 1500 }, + ), + ); + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 12000 }, + { cache_read: 0, cache_write: 0, output: 300, total: 4000 }, + ), + ); + expect(stripAnsi(text())).toBe(""); // no stats line is printed mid-turn + r.endTask(2345); + // Exact full-line assertion: context 4k (the latest request's total) and its delta, cumulative tokens 12k, + // per-task delta 5.5k (1500 + 4000), elapsed 2.3s (first task: session equals the delta); + // this also implies session_meta is not rendered (no /w or similar field appears in the output). + expect(stripAnsi(text())).toBe( + "[stats] context 4k (+4k) · tokens 12k (+5.5k) · 2.3s (+2.3s)\n", + ); + }); + + it("accumulates session elapsed across tasks; context delta is vs previous task", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + // task 1: context 4000, elapsed 2000ms. + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 4000 }, + { cache_read: 0, cache_write: 0, output: 0, total: 4000 }, + ), + ); + r.endTask(2000); + // task 2: context 7000 (+3000 vs. the previous task), session elapsed cumulative 5000ms (this task +3000ms). + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 11000 }, + { cache_read: 0, cache_write: 0, output: 0, total: 7000 }, + ), + ); + r.endTask(3000); + const lines = stripAnsi(text()).trim().split("\n"); + const last = lines[lines.length - 1]!; + // Exact full-line assertion: context 7k (delta = 7000 - 4000), cumulative session tokens 11k, + // per-task token delta 7k, total session elapsed 5s (this task +3s). + expect(last).toBe("[stats] context 7k (+3k) · tokens 11k (+7k) · 5s (+3s)"); + }); + + it("context delta goes negative after compaction shrinks the context (no clamping)", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + // task 1: context 7000. + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 7000 }, + { cache_read: 0, cache_write: 0, output: 0, total: 7000 }, + ), + ); + r.endTask(1000); + // task 2: context drops to 2000 after compaction -> delta is negative (2000 - 7000 = -5k), not clamped to non-negative. + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 9000 }, + { cache_read: 0, cache_write: 0, output: 0, total: 2000 }, + ), + ); + r.endTask(1000); + const lines = stripAnsi(text()).trim().split("\n"); + expect(lines[lines.length - 1]).toBe("[stats] context 2k (-5k) · tokens 9k (+2k) · 2s (+1s)"); + }); + + it("renders mode-specific compaction messages (summarize vs discard)", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(compactionBegin({ reason: "context", mode: "summarize", context: 150, turns: 3 })); + r.handle(compactionEnd({ reason: "context", mode: "summarize", status: "completed" })); + r.handle(compactionBegin({ reason: "manual", mode: "discard", context: 10, turns: 1 })); + r.handle(compactionEnd({ reason: "manual", mode: "discard", status: "completed" })); + r.handle(compactionEnd({ reason: "context", mode: "summarize", status: "failed" })); + expect(stripAnsi(text())).toBe( + [ + "[compaction] summarizing context (context)…", + "[compaction] done; continuing with the summarized context", + "[compaction] discarding context (manual)…", + "[compaction] done; old context discarded", + "[compaction] failed; keeping the current context", + "", + ].join("\n"), + ); + }); + + it("轮结束后的压缩:压缩完成行展示本次消耗,但不计入本轮统计增量;不更新上下文", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + // Ordinary request: context 5000. + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 8000 }, + { cache_read: 0, cache_write: 0, output: 0, total: 5000 }, + ), + ); + // The compaction request's usage sits between the paired compaction events: no ordinary request_end + // follows it in this turn -> compaction after the turn has ended. + r.handle(compactionBegin({ reason: "context", mode: "summarize", context: 5000, turns: 1 })); + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 14000 }, + { cache_read: 0, cache_write: 0, output: 0, total: 6000 }, + ), + ); + r.handle(compactionEnd({ reason: "context", mode: "summarize", status: "completed" })); + r.endTask(1000); + const s = stripAnsi(text()); + // The compaction-done line still shows this call's usage: session cumulative 14k + this compaction's 6k. + expect(s).toContain( + "[compaction] done; continuing with the summarized context · tokens 14k (+6k)", + ); + // Stats line: context stays at the ordinary-request figure of 5k; cumulative tokens 14k (includes + // compaction, following the provider), but this turn's **delta** is only the ordinary request's 5k — + // compaction after the turn ends is not attributed to this turn. + expect(s).toContain("context 5k"); + expect(s).toContain("tokens 14k (+5k)"); + }); + + it("轮途中的压缩(其后还有普通 request_end):用时含压缩跨度,Token 增量计入压缩", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + // own1: ordinary request, request 5000, 00:00 -> 00:02. + r.handle(at("2026-07-05T00:00:00.000Z", requestBegin())); + r.handle(at("2026-07-05T00:00:01.000Z", usage(5000, 5000))); + r.handle(at("2026-07-05T00:00:02.000Z", requestEnd("completed"))); + // Mid-turn compaction: 00:03 -> 00:13, request 6000 (the compaction's own summarization request). + r.handle( + at( + "2026-07-05T00:00:03.000Z", + compactionBegin({ reason: "context", mode: "summarize", context: 5000, turns: 1 }), + ), + ); + r.handle(at("2026-07-05T00:00:04.000Z", requestBegin())); + r.handle(at("2026-07-05T00:00:10.000Z", usage(6000, 14000))); + r.handle(at("2026-07-05T00:00:12.000Z", requestEnd("completed"))); + r.handle( + at( + "2026-07-05T00:00:13.000Z", + compactionEnd({ reason: "context", mode: "summarize", status: "completed" }), + ), + ); + // The turn continues after compaction (carry-over): own2 request 2000, final request_end at 00:16 -> settles the compaction usage. + r.handle(at("2026-07-05T00:00:14.000Z", requestBegin())); + r.handle(at("2026-07-05T00:00:15.000Z", usage(2000, 16000))); + r.handle(at("2026-07-05T00:00:16.000Z", requestEnd("completed"))); + r.endTask(999); // the passed-in wall clock is ignored: with a request_end present, elapsed comes from the timestamp span + const s = stripAnsi(text()); + // Elapsed = first event 00:00 -> the last non-compaction request_end 00:16 = 16s (includes the 10s of + // compaction in the middle, which occupied this turn's wall clock). + // Token delta = own1 5000 + own2 2000 + compaction 6000 = 13k; context uses the ordinary-request figure after compaction, 2k. + expect(s).toContain("context 2k"); + expect(s).toContain("tokens 16k (+13k)"); + expect(s).toContain("16s (+16s)"); + }); + + it("轮结束后的压缩(带 request 事件):用时止于压缩前的最后一个 request_end", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + // own1: 00:00 -> 00:03. + r.handle(at("2026-07-05T00:00:00.000Z", requestBegin())); + r.handle(at("2026-07-05T00:00:01.000Z", usage(5000, 5000))); + r.handle(at("2026-07-05T00:00:03.000Z", requestEnd("completed"))); + // Trailing compaction: 00:04 -> 00:24, a full 20s, with no ordinary request_end for this turn after it. + r.handle( + at( + "2026-07-05T00:00:04.000Z", + compactionBegin({ reason: "context", mode: "summarize", context: 5000, turns: 1 }), + ), + ); + r.handle(at("2026-07-05T00:00:05.000Z", requestBegin())); + r.handle(at("2026-07-05T00:00:20.000Z", usage(6000, 14000))); + r.handle(at("2026-07-05T00:00:23.000Z", requestEnd("completed"))); + r.handle( + at( + "2026-07-05T00:00:24.000Z", + compactionEnd({ reason: "context", mode: "summarize", status: "completed" }), + ), + ); + r.endTask(999); + const s = stripAnsi(text()); + // Elapsed = 00:00 -> the last non-compaction request_end before compaction, 00:03 = 3s (the whole 20s + // compaction span comes after it and does not count). + // Token delta is only own1's 5k; compaction's 6k is not attributed to this turn. + expect(s).toContain("tokens 14k (+5k)"); + expect(s).toContain("3s (+3s)"); + }); + + it("renders approval_decision events (approved / denied)", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(approvalDecision("allow", "c1")); + r.handle(approvalDecision("deny", "c2")); + const s = stripAnsi(text()); + expect(s).toContain("[approved]"); + expect(s).toContain("[denied]"); + }); + + it("keeps call → decision contiguous at prompt time and dedupes the late approval_decision event", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + const tc = toolCall({ name: "exec_command", arguments: '{"cmd":"pwd"}', toolCallId: "p8" }); + // Interactive approval: while locked, renders "call line -> (prompt, written directly by readline) -> result" as three contiguous lines. + r.beginUserPrompt(tc); + r.noteApprovalDecision(tc, "allow"); + r.endUserPrompt(); + expect(stripAnsi(text())).toBe("[tool-p8] $ pwd\n✓ [approved]\n"); + // A late approval_decision event is deduped by key and not re-rendered. + r.handle(approvalDecision("allow", "p8")); + expect(stripAnsi(text())).toBe("[tool-p8] $ pwd\n✓ [approved]\n"); + }); + + it("re-renders a half-streamed call line at approval and suppresses its late tail deltas", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + const tc = toolCall({ + name: "exec_command", + arguments: '{"cmd":"git status"}', + toolCallId: "h7", + }); + // The call line is still mid-stream (only half its arguments rendered) when approval begins. + r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "h7" })); + r.handle( + partialToolCall({ + eventType: "delta", + name: "", + arguments: '{"cmd":"git st', + toolCallId: "h7", + }), + ); + r.beginUserPrompt(tc); + // The trailing delta / stop arrive queued while the screen is locked. + r.handle( + partialToolCall({ eventType: "delta", name: "", arguments: 'atus"}', toolCallId: "h7" }), + ); + r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "h7" })); + r.noteApprovalDecision(tc, "allow"); + r.endUserPrompt(); + const s = stripAnsi(text()); + // At approval time, the full call line is re-rendered in place from the complete message, right next to + // the result; after unlocking, the late tail is deduped and must not start a duplicate call line after + // the result line. + expect(s).toContain("[tool-h7] $ git status\n✓ [approved]\n"); + expect(s.slice(s.indexOf("[approved]"))).not.toContain("[tool-h7]"); + }); + + it("defers another call's auto-approval rendering while an interactive prompt is active", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + const parent = toolCall({ + name: "exec_command", + arguments: '{"cmd":"pwd"}', + toolCallId: "pa1", + }); + const child = withOrigin( + toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "ch2" }), + "sess_kid", + ); + r.beginUserPrompt(parent); // parent call's interactive prompt: locks the screen + r.noteApprovalDecision(child, "allow"); // concurrent subagent auto-approval: deferred, not inserted mid-prompt + expect(stripAnsi(text())).not.toContain("ch2"); + r.noteApprovalDecision(parent, "allow"); // the prompt owner's result renders in place as usual + r.endUserPrompt(); + const s = stripAnsi(text()); + // Order: parent call line -> parent result -> child call line -> child result. + const iParentOk = s.indexOf("[approved]"); + const iChildCall = s.indexOf("[agent-kid-tool-ch2]"); + expect(s.indexOf("[tool-pa1]")).toBeGreaterThanOrEqual(0); + expect(iChildCall).toBeGreaterThan(iParentOk); + expect(s.indexOf("[approved]", iChildCall)).toBeGreaterThan(iChildCall); + }); + + it("endCompact settles manual /compact usage so the next task's delta excludes it", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 8000 }, + { cache_read: 0, cache_write: 0, output: 0, total: 5000 }, + ), + ); + r.endTask(1000); + // Manual /compact: the compaction request consumes 6000 (already shown on the compaction-done line), endCompact settles it. + r.handle(compactionBegin({ reason: "manual", mode: "summarize", context: 5000, turns: 1 })); + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 14000 }, + { cache_read: 0, cache_write: 0, output: 0, total: 6000 }, + ), + ); + r.handle(compactionEnd({ reason: "manual", mode: "summarize", status: "completed" })); + r.endCompact(500); + // The next task consumes only 1000: its delta must not include compaction's 6000. + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 15000 }, + { cache_read: 0, cache_write: 0, output: 0, total: 1000 }, + ), + ); + r.endTask(1000); + const lines = stripAnsi(text()).trim().split("\n"); + expect(lines[lines.length - 1]).toContain("tokens 15k (+1k)"); + }); + + it("re-renders the call line next to the decision when other output separated them (auto-approve)", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + // The call line is first rendered while streaming, then separated from the decision by other output. + r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "c5" })); + r.handle( + partialToolCall({ + eventType: "delta", + name: "", + arguments: '{"cmd":"ls"}', + toolCallId: "c5", + }), + ); + r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "c5" })); + r.handle(partialText("start", "")); + r.handle(partialText("delta", "hi")); + r.handle(partialText("stop", "", "completed")); + // Auto-approval: the call line is no longer adjacent -> it is re-rendered in place, with the result immediately following it as a pair. + r.noteApprovalDecision( + toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c5" }), + "allow", + ); + expect(stripAnsi(text())).toBe("[tool-c5] $ ls\nhi\n[tool-c5] $ ls\n✓ [approved]\n"); + }); + + it("does not re-render the call line when it is already adjacent to the decision", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "c6" })); + r.handle( + partialToolCall({ + eventType: "delta", + name: "", + arguments: '{"cmd":"ls"}', + toolCallId: "c6", + }), + ); + r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "c6" })); + r.noteApprovalDecision( + toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c6" }), + "deny", + ); + expect(stripAnsi(text())).toBe("[tool-c6] $ ls\n× [denied]\n"); + }); +}); + +describe("StreamRenderer — nested (origin-tagged) subagent messages", () => { + const hop: MessageOrigin = "sess_child"; + + it("renders nested tool calls with an agent-tool tag; skips nested text/thinking partials", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + // Nested text/thinking is not rendered (the child's reply is shown via the parent tool's output gutter). + r.handle(withOrigin(partialText("delta", "child text"), hop)); + r.handle(withOrigin(partialThinking("delta", "child think"), hop)); + // A nested complete tool_call renders one line (so the user can see what tool the subagent is calling + // before approval); the tag is agent--tool-; the + // approval line carries no tag. + r.handle( + withOrigin( + toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "cc1" }), + hop, + ), + ); + r.handle(withOrigin(approvalDecision("allow", "cc1"), hop)); + expect(stripAnsi(text())).toBe("[agent-ild-tool-cc1] $ ls\n✓ [approved]\n"); + }); + + it("renders the pending nested tool call at approval time when its stream copy has not arrived; dedupes the late copy", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + const tc = withOrigin( + toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "cc9" }), + hop, + ); + // The approval callback arrives before the forwarded message: beginUserPrompt renders the call line directly from the complete message. + r.beginUserPrompt(tc); + expect(stripAnsi(text())).toBe("[agent-ild-tool-cc9] $ ls\n"); + r.endUserPrompt(); + // The late forwarded copy is deduped by key and not re-rendered. + r.handle(tc); + expect(stripAnsi(text())).toBe("[agent-ild-tool-cc9] $ ls\n"); + }); + + it("renders the pending parent tool call at approval time and suppresses its late partial stream", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.beginUserPrompt( + toolCall({ name: "exec_command", arguments: '{"cmd":"pwd"}', toolCallId: "p7" }), + ); + r.endUserPrompt(); + // The whole late streaming copy is deduped and skipped. + r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "p7" })); + r.handle( + partialToolCall({ + eventType: "delta", + name: "", + arguments: '{"cmd":"pwd"}', + toolCallId: "p7", + }), + ); + r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "p7" })); + expect(stripAnsi(text())).toBe("[tool-p7] $ pwd\n"); + }); + + it("adds nested token_usage request totals to the task delta and the session total", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + // One parent-session request: 1500; one child-session request: 2000 -> per-task delta 3.5k; + // session cumulative = parent 8000 + child 2000 = 10k (delta and cumulative use the same basis: parent + child). + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 8000 }, + { cache_read: 0, cache_write: 0, output: 200, total: 1500 }, + ), + ); + r.handle( + withOrigin( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 2000 }, + { cache_read: 0, cache_write: 0, output: 100, total: 2000 }, + ), + hop, + ), + ); + r.endTask(1000); + const s1 = stripAnsi(text()); + expect(s1).toContain("3.5k"); // the per-task delta includes child-session usage + expect(s1).toContain("10k"); // the session cumulative includes child-session usage + // The child session's cumulative persists across tasks: the next task consumes only from the parent session, cumulative = 9000 + 2000 = 11k (+1k). + r.handle( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 9000 }, + { cache_read: 0, cache_write: 0, output: 100, total: 1000 }, + ), + ); + r.endTask(1000); + const lines = stripAnsi(text()).trim().split("\n"); + const last = lines[lines.length - 1]!; + expect(last).toContain("11k"); + expect(last).toContain("+1k"); + }); + + it("prints stats when a task only has nested (subagent) token usage", () => { + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle( + withOrigin( + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 2000 }, + { cache_read: 0, cache_write: 0, output: 100, total: 2000 }, + ), + hop, + ), + ); + r.endTask(1000); + const s = stripAnsi(text()); + expect(s).toContain("[stats]"); + expect(s).toContain("2k (+2k)"); + }); +}); + +describe("renderHistory (resume)", () => { + it("renders complete messages statically with interruption markers", async () => { + const { renderHistory } = await import("../src/render.js"); + const { userText } = await import("@prismshadow/penguin-core"); + const { stream, text } = collector(); + renderHistory( + [ + userText("hello"), + thinkingMessage("pondering"), + assistantText("hi there"), + toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "call_653" }), + toolCallOutput({ output: "a.txt\nb.txt", toolCallId: "call_653" }), + assistantText("half answer", "aborted"), + ], + stream, + ); + const s = stripAnsi(text()); + expect(s).toContain("> hello"); + expect(s).toContain("pondering"); + expect(s).toContain("hi there"); + expect(s).toContain("[tool-653] $ ls"); + expect(s).toContain("[tool-653] >> a.txt"); + expect(s).toContain("[tool-653] >> b.txt"); + // An interrupted message carries a marker (rendering includes the interrupted turn). + expect(s).toContain("half answer [aborted]"); + }); + + it("skips events and renders nothing for empty history", async () => { + const { renderHistory } = await import("../src/render.js"); + const { stream, text } = collector(); + renderHistory( + [ + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: 1 }, + { cache_read: 0, cache_write: 0, output: 0, total: 1 }, + ), + ], + stream, + ); + expect(text()).toBe(""); + }); +}); diff --git a/packages/cli/test/serve.test.ts b/packages/cli/test/serve.test.ts new file mode 100644 index 0000000..07ab3f3 --- /dev/null +++ b/packages/cli/test/serve.test.ts @@ -0,0 +1,72 @@ +import { describe, expect, it } from "vitest"; +import { Command } from "commander"; +import { + DEFAULT_HOST, + DEFAULT_PORT, + browserCommand, + browserUrl, + registerServeCommands, + resolvePort, +} from "../src/commands/serve.js"; +import { getMessages } from "../src/i18n.js"; + +describe("resolvePort(选项 > 环境变量 > 缺省 7364)", () => { + it("都未给时用缺省 7364", () => { + expect(DEFAULT_PORT).toBe(7364); + expect(resolvePort(undefined, undefined)).toBe(7364); + expect(resolvePort(undefined, "")).toBe(7364); // an empty string counts as unset + }); + it("只有环境变量时取环境变量", () => { + expect(resolvePort(undefined, "8080")).toBe(8080); + }); + it("选项优先于环境变量", () => { + expect(resolvePort("9000", "8080")).toBe(9000); + }); + it("非法值(非整数 / 越界)抛错", () => { + expect(() => resolvePort("abc", undefined)).toThrow(/abc/); + expect(() => resolvePort("3.14", undefined)).toThrow(); + expect(() => resolvePort("-1", undefined)).toThrow(); + expect(() => resolvePort("65536", undefined)).toThrow(); + expect(() => resolvePort(undefined, "not-a-port")).toThrow(/not-a-port/); + }); +}); + +describe("browserCommand(按平台选择打开命令)", () => { + const url = "http://127.0.0.1:7364/"; + it("darwin → open", () => { + expect(browserCommand("darwin", url)).toEqual({ command: "open", args: [url] }); + }); + it("win32 → cmd /c start(空标题占位在 URL 前)", () => { + expect(browserCommand("win32", url)).toEqual({ + command: "cmd", + args: ["/c", "start", "", url], + }); + }); + it("其他平台(linux 等)→ xdg-open", () => { + expect(browserCommand("linux", url)).toEqual({ command: "xdg-open", args: [url] }); + expect(browserCommand("freebsd", url)).toEqual({ command: "xdg-open", args: [url] }); + }); +}); + +describe("browserUrl(通配监听地址转 127.0.0.1)", () => { + it("常规 host 原样拼接", () => { + expect(browserUrl(DEFAULT_HOST, 7364)).toBe("http://127.0.0.1:7364/"); + expect(browserUrl("192.168.1.2", 8080)).toBe("http://192.168.1.2:8080/"); + }); + it("0.0.0.0 / :: 时浏览器 URL 用 127.0.0.1", () => { + expect(browserUrl("0.0.0.0", 7364)).toBe("http://127.0.0.1:7364/"); + expect(browserUrl("::", 7364)).toBe("http://127.0.0.1:7364/"); + }); +}); + +describe("registerServeCommands(命令注册)", () => { + it("注册 server 与 web 两个顶层命令,web 缺省 open=true(--no-open 可关)", () => { + const program = new Command(); + registerServeCommands(program, getMessages("en")); + const names = program.commands.map((c) => c.name()); + expect(names).toContain("server"); + expect(names).toContain("web"); + const web = program.commands.find((c) => c.name() === "web")!; + expect(web.opts().open).toBe(true); + }); +}); diff --git a/packages/cli/test/task-loop.test.ts b/packages/cli/test/task-loop.test.ts new file mode 100644 index 0000000..61751d1 --- /dev/null +++ b/packages/cli/test/task-loop.test.ts @@ -0,0 +1,51 @@ +/** + * runTask's result reporting: when a Task ends with a main-session abort event (LLM + * failure / reconnect exhausted / user interrupt), it reports aborted=true, which + * `penguin run` maps to a non-zero exit code; a sub-session abort does not count. + */ +import { describe, expect, it } from "vitest"; +import { Writable } from "node:stream"; +import { abortEvent, assistantText, withOrigin } from "@prismshadow/penguin-core"; +import type { OmniMessage, Session } from "@prismshadow/penguin-core"; +import { StreamRenderer } from "../src/render.js"; +import { runTask } from "../src/task-loop.js"; +import { getMessages } from "../src/i18n.js"; + +const t = getMessages("en"); + +function fakeSession(messages: OmniMessage[]): Session { + return { + async *run() { + for (const m of messages) yield m; + }, + toolPermission: () => "rw", + } as unknown as Session; +} + +function silentRenderer(): StreamRenderer { + const stream = new Writable({ + write(_chunk, _enc, cb) { + cb(); + }, + }); + return new StreamRenderer(stream, t); +} + +describe("runTask abort reporting", () => { + it("reports aborted=true when the task ends with a main-session abort event", async () => { + const result = await runTask(fakeSession([abortEvent("llm request error: 401")]), [], { + renderer: silentRenderer(), + t, + }); + expect(result.aborted).toBe(true); + }); + + it("reports aborted=false on normal completion; child-session aborts do not count", async () => { + const result = await runTask( + fakeSession([withOrigin(abortEvent("child aborted"), "sess_child"), assistantText("done")]), + [], + { renderer: silentRenderer(), t }, + ); + expect(result.aborted).toBe(false); + }); +}); diff --git a/packages/cli/test/tool-render.test.ts b/packages/cli/test/tool-render.test.ts new file mode 100644 index 0000000..9da989b --- /dev/null +++ b/packages/cli/test/tool-render.test.ts @@ -0,0 +1,89 @@ +import { describe, expect, it } from "vitest"; +import { renderPartialToolCall } from "../src/tool-render.js"; + +describe("renderPartialToolCall", () => { + it("renders partial exec_command args as $ ", () => { + expect(renderPartialToolCall("exec_command", '{"cmd":')).toBeNull(); + expect(renderPartialToolCall("exec_command", '{"cmd":"l')).toBe("$ l"); + expect(renderPartialToolCall("exec_command", '{"cmd":"ls"}')).toBe("$ ls"); + expect(renderPartialToolCall("exec_command", '{"cmd":"echo \\"hi\\"')).toBe('$ echo "hi"'); + }); + + it("renders run_subagent as run_subagent << , folded to one line", () => { + expect(renderPartialToolCall("run_subagent", '{"prompt":')).toBeNull(); + expect(renderPartialToolCall("run_subagent", '{"prompt":"analy')).toBe("run_subagent << analy"); + expect(renderPartialToolCall("run_subagent", '{"prompt":"line1\\nline2"}')).toBe( + "run_subagent << line1 line2", + ); + }); + + it("renders input_command polls (empty chars) without a payload", () => { + expect(renderPartialToolCall("input_command", '{"process_id":')).toBeNull(); + expect(renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d"}')).toBe( + "⌨ input_command → proc-1a2b3c4d", + ); + expect( + renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":""}'), + ).toBe("⌨ input_command → proc-1a2b3c4d"); + }); + + it("renders non-empty input_command chars with visible control characters", () => { + expect( + renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":"y\\n"}'), + ).toBe("⌨ input_command → proc-1a2b3c4d << y\\n"); + // U+0003 (Ctrl-C) is rendered in caret notation. + expect( + renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":"\\u0003"}'), + ).toBe("⌨ input_command → proc-1a2b3c4d << ^C"); + // Disambiguates literal backslash escapes: chars "a", "\", "n" render as a\\n, distinct from a real newline \n. + expect( + renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":"a\\\\n"}'), + ).toBe("⌨ input_command → proc-1a2b3c4d << a\\\\n"); + }); + + it("keeps input_command previews append-only across \\uXXXX delta boundaries", () => { + const stages = [ + '{"process_id":"proc-1a2b3c4d","chars":"y', + '{"process_id":"proc-1a2b3c4d","chars":"y\\u0', + '{"process_id":"proc-1a2b3c4d","chars":"y\\u0003', + ]; + const previews = stages.map((s) => renderPartialToolCall("input_command", s)!); + expect(previews[0]).toBe("⌨ input_command → proc-1a2b3c4d << y"); + // An incomplete \u escape is treated as "stop here" rather than emitting the raw hex as literal text. + expect(previews[1]).toBe("⌨ input_command → proc-1a2b3c4d << y"); + expect(previews[2]).toBe("⌨ input_command → proc-1a2b3c4d << y^C"); + for (let i = 1; i < previews.length; i++) { + expect(previews[i]!.startsWith(previews[i - 1]!)).toBe(true); + } + }); + + it("renders input_subagent polls without a payload and follow-up prompts with one", () => { + expect( + renderPartialToolCall("input_subagent", '{"subagent_id":"subagent-9f8e7d6c","prompt":""}'), + ).toBe("⌨ input_subagent → subagent-9f8e7d6c"); + expect( + renderPartialToolCall( + "input_subagent", + '{"subagent_id":"subagent-9f8e7d6c","prompt":"continue with the tests"}', + ), + ).toBe("⌨ input_subagent → subagent-9f8e7d6c << continue with the tests"); + }); + + it("truncates long payload previews and stops growing afterwards", () => { + const long = "x".repeat(130); + const capped = renderPartialToolCall( + "input_subagent", + `{"subagent_id":"subagent-9f8e7d6c","prompt":"${long}"}`, + ); + expect(capped).toBe(`⌨ input_subagent → subagent-9f8e7d6c << ${"x".repeat(120)}…`); + const longer = renderPartialToolCall( + "input_subagent", + `{"subagent_id":"subagent-9f8e7d6c","prompt":"${long}yyy"}`, + ); + expect(longer).toBe(capped); + }); + + it("falls back to name(args-prefix) for unknown tools", () => { + expect(renderPartialToolCall("search", '{"q":"hi')).toBe('search({"q":"hi'); + }); +}); diff --git a/packages/cli/tsconfig.json b/packages/cli/tsconfig.json new file mode 100644 index 0000000..8cd1715 --- /dev/null +++ b/packages/cli/tsconfig.json @@ -0,0 +1,7 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { + "rootDir": "." + }, + "include": ["src", "test"] +} diff --git a/packages/cli/tsup.config.ts b/packages/cli/tsup.config.ts new file mode 100644 index 0000000..acc70b4 --- /dev/null +++ b/packages/cli/tsup.config.ts @@ -0,0 +1,22 @@ +import { defineConfig } from "tsup"; + +export default defineConfig({ + entry: ["src/index.ts"], + format: ["esm"], + target: "node20", + platform: "node", + clean: true, + sourcemap: true, + // Bundle the workspace source @prismshadow/penguin-core, but keep third-party deps (incl. + // CJS yaml / smol-toml / agenthub) external and resolved from node_modules at runtime — + // avoids bundling CJS deps into ESM and triggering a "Dynamic require" error. + // @prismshadow/penguin-skills must stay external: it reads files under its own skills/ dir + // at runtime (files are the source of truth); bundling would break paths relative to the + // package root. cli already declares it as a direct dependency. + // @prismshadow/penguin-server stays external: the penguin server / web commands import it + // dynamically at runtime. + // tsup treats this package's package.json dependencies as external by default, so these + // deps are already declared there. + noExternal: ["@prismshadow/penguin-core"], + banner: { js: "#!/usr/bin/env node" }, +}); diff --git a/packages/core/README.md b/packages/core/README.md new file mode 100644 index 0000000..f8b1dcf --- /dev/null +++ b/packages/core/README.md @@ -0,0 +1,41 @@ +# @prismshadow/penguin-core + +The PenguinHarness SDK and execution engine: the ReAct loop (`context_engine`), the OmniMessage protocol, the LLM / Environment interface contracts, Agent State and append-only Traces. + +The engine speaks only OmniMessage and delegates everything else through two swappable interfaces — `LLMInterface` (models, via the [`@prismshadow/agenthub`](https://www.npmjs.com/package/@prismshadow/agenthub) gateway) and `EnvironmentInterface` (tool execution). The SDK caller is the Human boundary: one entry point, `session.run`, streams the whole loop. + +```ts +import { createAgent, isCompleteModelMessage, userText } from "@prismshadow/penguin-core"; + +const agent = await createAgent({ agentId: "default_agent" }); +const session = await agent.createSession({ workspaceDir: process.cwd() }); + +for await (const output of session.run([userText("Create hello.txt containing hi")], { + approve: async () => "allow", // per-tool-call approval +})) { + if (isCompleteModelMessage(output) && output.payload.type === "text") { + console.log(output.payload.text); + } +} +``` + +A single `run` drives a complete Task: streaming output, per-call approvals, concurrent tool execution, interrupt carry-over, automatic reconnect and context compaction. State lives under `~/.penguin/data` (`PENGUIN_HOME`); every Session restores fully from its Trace. + +## Documentation + +- [Architecture](https://prism-shadow.github.io/penguin-harness/docs/architecture) +- [The OmniMessage Protocol](https://prism-shadow.github.io/penguin-harness/docs/omni-message) +- [Core Interfaces](https://prism-shadow.github.io/penguin-harness/docs/interfaces) +- [The Agent Loop](https://prism-shadow.github.io/penguin-harness/docs/agent-loop) +- [Sessions & Traces](https://prism-shadow.github.io/penguin-harness/docs/sessions-and-traces) + +## Development + +```bash +pnpm --filter @prismshadow/penguin-core build # tsup → dist/ (exports point at dist) +pnpm --filter @prismshadow/penguin-core typecheck +pnpm --filter @prismshadow/penguin-core test +pnpm test:e2e # live-model e2e (needs DEEPSEEK_API_KEY) +``` + +Part of [PenguinHarness](https://github.com/Prism-Shadow/penguin-harness) · Apache-2.0 diff --git a/packages/core/package.json b/packages/core/package.json new file mode 100644 index 0000000..593540e --- /dev/null +++ b/packages/core/package.json @@ -0,0 +1,58 @@ +{ + "name": "@prismshadow/penguin-core", + "version": "0.0.1", + "type": "module", + "description": "PenguinHarness core SDK: context_engine, OmniMessage protocol, LLM/Environment interfaces.", + "license": "Apache-2.0", + "repository": { + "type": "git", + "url": "git+https://github.com/Prism-Shadow/penguin-harness.git", + "directory": "packages/core" + }, + "exports": { + ".": { + "types": "./dist/index.d.ts", + "import": "./dist/index.js" + }, + "./omnimessage": { + "types": "./dist/omnimessage/index.d.ts", + "import": "./dist/omnimessage/index.js" + }, + "./interfaces": { + "types": "./dist/interfaces.d.ts", + "import": "./dist/interfaces.js" + }, + "./model-catalog": { + "types": "./dist/state/model-catalog.d.ts", + "import": "./dist/state/model-catalog.js" + } + }, + "main": "./dist/index.js", + "types": "./dist/index.d.ts", + "scripts": { + "typecheck": "tsc --noEmit -p tsconfig.json", + "test": "vitest run --passWithNoTests", + "test:e2e": "PENGUIN_E2E=1 vitest run test/llm.e2e.test.ts", + "build": "tsup" + }, + "dependencies": { + "@prismshadow/agenthub": "^0.3.3", + "@prismshadow/penguin-skills": "workspace:*", + "smol-toml": "^1.3.0", + "yaml": "^2.5.0" + }, + "devDependencies": { + "@types/node": "^24.0.0", + "dotenv": "^17.0.0", + "tsup": "^8.3.0", + "typescript": "^5.6.0", + "vitest": "^2.1.0" + }, + "files": [ + "dist", + "LICENSE" + ], + "publishConfig": { + "access": "public" + } +} diff --git a/packages/core/src/agent.ts b/packages/core/src/agent.ts new file mode 100644 index 0000000..167c87a --- /dev/null +++ b/packages/core/src/agent.ts @@ -0,0 +1,643 @@ +/** + * Agent and the `createAgent` entry point. + * + * `createAgent` is the unified way to create/load an Agent: it initializes Agent State + * if the directory is empty, otherwise loads by agentId. + * An Agent has exactly one Agent State and can run multiple times; the + * Workspace is determined when a Session is created. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { + assertValidId, + assembleSystemPrompt, + buildToolConfig, + selectBuiltinToolsForModel, + DEFAULT_COMPACTION_PROMPT, + formatModelRef, + getModel, + listInstalledSkills, + loadAgentVault, + loadOrInitAgentState, + loadProjectConfig, + projectDir, + resolveModelRef, + scratchpadDir, + systemConfigPath, + tracesDir, + type AgentState, + type ModelRef, + type ProjectConfig, +} from "./state/index.js"; +import { GenerativeModel, ToolCallIdAllocator } from "./llm/index.js"; +import { Environment } from "./environment/index.js"; +import { + Writer, + findLatestTraceFile, + latestSessionId as latestTraceSessionId, + readTraceTolerant, + resumeTrace, +} from "./trace/index.js"; +import { Session } from "./session.js"; +import { + createTempWorkspace, + formatSessionId, + sessionEnvironment, +} from "./internal/session-support.js"; +import { userText, withOrigin } from "./omnimessage/index.js"; +import type { + MessageOrigin, + OmniMessage, + TokenCounts, + ToolCallPayload, +} from "./omnimessage/index.js"; +import { SUBAGENT_NAME } from "./environment/tools/run-subagent.js"; +import { INPUT_SUBAGENT_NAME } from "./environment/tools/input-subagent.js"; +import type { CompactionSettings } from "./engine/context-engine.js"; +import type { + GenerativeModelConfig, + SubagentRunner, + ToolDefinition, + VisionDescriberService, +} from "./interfaces.js"; +import type { ModelEntry } from "./state/index.js"; + +/** + * Maximum subagent spawn depth. Currently capped at 1 level (a subagent cannot spawn + * another subagent); the depth mechanism is designed to support multiple levels — + * raise this constant to allow deeper nesting. + */ +const MAX_SUBAGENT_DEPTH = 1; + +export interface CreateAgentOptions { + agentId?: string; + projectId?: string; + /** Local data root directory; defaults to `resolveRoot()` (PENGUIN_HOME or ~/.penguin/data). */ + root?: string; +} + +export interface CreateSessionOptions { + /** Workspace for this run; if unspecified, a temporary Workspace is created under the Agent directory. */ + workspaceDir?: string; + /** Model used for this Session (upstream model_id); if unspecified, uses the Project's default Model. */ + modelId?: string; + /** + * Provider grouping for `modelId` (a paired reference); if omitted, resolved via + * `resolveModelRef` semantics — `model_id` only resolves if it is a globally unique + * exact match in the config; zero or multiple matches produce a clear error. + */ + provider?: string; + /** Explicit credentials; if unspecified, falls back to credentials in the Project config, then to AgentHub reading environment variables. */ + apiKey?: string; + baseUrl?: string; + /** Internal use: this Session's depth in the subagent spawn chain (0 at the top level), used to cap spawn depth. */ + subagentDepth?: number; +} + +export interface ResumeSessionOptions { + /** Id of the Session to resume. */ + sessionId: string; + /** Explicit credentials; if unspecified, falls back to credentials in the Project config, then to AgentHub reading environment variables. */ + apiKey?: string; + baseUrl?: string; +} + +/** + * Effective compaction threshold: capped at 75% of the model's `context_window` — + * the threshold must stay well below the hard window limit, otherwise small-window + * models get rejected by the provider (a non-retryable 400) before compaction even + * triggers, and the compaction request itself (old context + prompt + summary output) + * also needs headroom. Not clamped when `<=0` (disabled) or the window is unknown. + */ +export function effectiveMaxContextLength(configured: number, contextWindow: unknown): number { + if (configured <= 0) return configured; + if (typeof contextWindow !== "number") return configured; + return Math.min(configured, Math.floor(contextWindow * 0.75)); +} + +/** Create or load an Agent. */ +export async function createAgent(opts: CreateAgentOptions = {}): Promise { + const state = await loadOrInitAgentState(opts); + const projectConfig = await loadProjectConfig(state.root, state.projectId); + return new Agent(state, projectConfig); +} + +export class Agent { + constructor( + readonly state: AgentState, + readonly projectConfig: ProjectConfig, + ) {} + + /** + * Create a Session in the specified (or a temporary) Workspace. + * Docs: /docs/sessions-and-traces § "Run model". + */ + async createSession(opts: CreateSessionOptions = {}): Promise { + // Model is validated first (before creating the Workspace, so failure leaves no + // temp directory behind): the reference must resolve to an entry in the Project + // config (the (provider, model_id) pair is the unique key); a reference + // outside the config throws immediately rather than passing silently — otherwise + // credentials, pricing, and the context window would all be unavailable. + if (opts.modelId === undefined && opts.provider !== undefined) { + throw new Error( + "指定了 provider 却未指定 modelId:模型引用须成对给出(provider 不能单独使用)。", + ); + } + let ref: ModelRef; + if (opts.modelId !== undefined) { + // The only entry point for resolving an "omitted provider" reference (resolveModelRef): three branches — unique match / zero matches / ambiguous. + ref = resolveModelRef(this.projectConfig, opts.modelId, opts.provider); + } else if (this.projectConfig.default_model) { + ref = this.projectConfig.default_model; + } else { + throw new Error( + "未指定 modelId,且 Project 配置中没有 default_model。请用 `penguin config model add/default` 设置默认模型。", + ); + } + const modelEntry = getModel(this.projectConfig, ref); + if (!modelEntry) { + throw new Error( + `Model 不在 Project 配置中:${formatModelRef(ref)}。请用 \`penguin config model list\` 查看已配置的模型,或用 \`penguin config model add\` 添加。`, + ); + } + // Credentials are inlined on the model entry (single config file); an + // explicit argument takes priority, falling back to AgentHub reading env vars + // when both are absent. + const apiKey = opts.apiKey ?? modelEntry.api_key; + const baseUrl = opts.baseUrl ?? modelEntry.base_url; + + // An explicit Workspace must already exist as a directory: if it + // doesn't, throw rather than auto-create (to avoid a typo silently working in + // the wrong location); a temp Workspace is only created when unspecified. + let workspaceDir: string; + if (opts.workspaceDir) { + workspaceDir = path.resolve(opts.workspaceDir); + let stat; + try { + stat = await fs.stat(workspaceDir); + } catch { + throw new Error( + `Workspace 不存在:${workspaceDir}。请指定一个已存在的目录,或不指定 Workspace 以使用临时目录。`, + ); + } + if (!stat.isDirectory()) { + throw new Error(`Workspace 不是目录:${workspaceDir}。`); + } + } else { + workspaceDir = await createTempWorkspace( + this.state.root, + this.state.projectId, + this.state.agentId, + ); + } + const sessionId = formatSessionId(); + const subagentDepth = opts.subagentDepth ?? 0; + + // Agent-level vault (agent_state/.vault.toml) and installed Skills: read the current values each time a Session is created. + const vault = await loadAgentVault(this.state.root, this.state.projectId, this.state.agentId); + const installedSkills = await listInstalledSkills( + this.state.root, + this.state.projectId, + this.state.agentId, + ); + + // The assembled system prompt goes both to the LLM and into session_meta (so the + // Trace can audit the actual effective value). The vault only injects **key names** + // into the prompt (so the model knows which API keys are available); values only + // go into the subprocess environment. Skills only inject metadata (name and + // description); the model reads the body on demand via shell. + const systemPrompt = assembleSystemPrompt( + this.state, + sessionEnvironment(workspaceDir, sessionId, { + agentId: this.state.agentId, + projectDir: projectDir(this.state.root, this.state.projectId), + }), + Object.keys(vault), + installedSkills, + ); + + const rt = await this.buildRuntime({ + workspaceDir, + modelEntry, + apiKey, + baseUrl, + systemPrompt, + subagentDepth, + vault, + }); + + const trace = new Writer({ + tracesDir: tracesDir(this.state.root, this.state.projectId, this.state.agentId), + sessionId, + }); + + return new Session({ + meta: { + session_id: sessionId, + provider: modelEntry.provider, + model_id: modelEntry.model_id, + model_context_window: modelEntry.context_window ?? "unknown", + system_prompt: systemPrompt, + tools: rt.tools, + thinking_level: this.state.systemConfig.model?.thinking_level ?? "default", + agent_state: this.state.stateDir, + workspace: workspaceDir, + }, + llm: rt.llm, + environment: rt.environment, + trace, + createLLM: rt.createLLM, + createBareLLM: rt.createBareLLM, + compaction: rt.compaction, + // Model doesn't support images: input images are written to the session scratchpad and their paths appended to the text (viewed via describe_image). + ...(modelEntry.vision === false + ? { + inputImagesDir: path.join( + scratchpadDir(this.state.root, this.state.projectId, this.state.agentId), + sessionId, + ), + } + : {}), + // Max turns comes from the Agent's system_config (runtime parameters belong to the Agent config). + ...(this.state.systemConfig.max_turns !== undefined + ? { maxTurns: this.state.systemConfig.max_turns } + : {}), + }); + } + + /** + * Resume an existing Session and continue the conversation. + * + * The resume source is the Session's **latest-index** Trace file: runtime config is + * read from its `session_meta` (Model, the original system prompt text, and the + * Workspace all carry over from the original Session and cannot be changed), while + * tools and Environment are reassembled from the current Agent State. The replayed, + * already-committed history is injected once via AgentHub's setHistory (used only on + * resume); any leftover input is rebuilt as carry-over (paired fallback placeholders + * are synthesized in memory only, never written to the Trace). Messages after resume + * continue in the original Trace file (the file follows the context, not the date), + * and Token / turn-count stats carry over from their original values. + * Docs: /docs/sessions-and-traces § "Session recovery". + */ + async resumeSession(opts: ResumeSessionOptions): Promise { + const { sessionId } = opts; + const dir = tracesDir(this.state.root, this.state.projectId, this.state.agentId); + const located = await findLatestTraceFile(dir, sessionId); + if (!located) { + throw new Error(`Session 不存在:${sessionId}(在 ${dir} 下未找到对应 Trace 文件)。`); + } + const resumed = resumeTrace(await readTraceTolerant(located.path)); + if (!resumed.meta) { + throw new Error(`Trace 缺少 session_meta,无法恢复:${located.path}`); + } + const meta = resumed.meta.payload; + // Model reference is stored as a pair in session_meta; a missing provider means legacy data (no migration since the product hasn't shipped yet). + if (typeof meta.provider !== "string") { + throw new Error( + `Trace 来自旧版本数据(session_meta 缺少 provider,模型引用未分列):${located.path}。请删除数据目录后重建会话。`, + ); + } + + // The Workspace carries over from the original Session and must still exist (throw if missing, never auto-create). + const workspaceDir = meta.workspace; + let stat; + try { + stat = await fs.stat(workspaceDir); + } catch { + throw new Error(`原 Session 的 Workspace 已不存在:${workspaceDir},无法恢复。`); + } + if (!stat.isDirectory()) { + throw new Error(`原 Session 的 Workspace 不是目录:${workspaceDir},无法恢复。`); + } + + // The Model carries over from the original Session (paired reference) and must still be present in the Project config. + const ref: ModelRef = { provider: meta.provider, model_id: meta.model_id }; + const modelEntry = getModel(this.projectConfig, ref); + if (!modelEntry) { + throw new Error( + `原 Session 的 Model 不在 Project 配置中:${formatModelRef(ref)}。请用 \`penguin config model add\` 重新配置后再恢复。`, + ); + } + const apiKey = opts.apiKey ?? modelEntry.api_key; + const baseUrl = opts.baseUrl ?? modelEntry.base_url; + + // Tools and Environment are reassembled from the current Agent State (tool + // definitions are passed with every Request and aren't part of the history); the + // system prompt uses the original text recorded in the Trace (identical to the + // original history); the vault uses current values (it's injected into the + // subprocess environment, not the history, so a resumed Session should get the + // latest keys too). + const rt = await this.buildRuntime({ + workspaceDir, + modelEntry, + apiKey, + baseUrl, + systemPrompt: meta.system_prompt, + subagentDepth: 0, + vault: await loadAgentVault(this.state.root, this.state.projectId, this.state.agentId), + }); + + // History is injected once into a fresh context object (setHistory is only used + // on resume); Session cumulative Token counts carry over. Wrap the error + // descriptively: bad tool arguments in the history (e.g. truncated JSON written by + // a third-party OpenAI-compatible endpoint) throw a raw SyntaxError during + // conversion, so the error must indicate Trace history corruption rather than a + // regular runtime error. + if (resumed.history.length > 0) { + try { + rt.llm.setHistory(resumed.history); + } catch (err) { + const detail = err instanceof Error ? err.message : String(err); + throw new Error( + `恢复失败:Trace 历史无法注入(记录可能损坏,如非法的工具参数 JSON):${detail}`, + ); + } + } + rt.llm.sessionTokens = resumed.sessionTokens; + + // Continue writing to the original Trace file (the Trace only records real messages; synthesized paired placeholders are re-emitted in memory alongside carry-over). + const trace = new Writer({ + tracesDir: dir, + sessionId, + dateDir: located.dateDir, + startIndex: located.index, + }); + + return new Session({ + meta: { + session_id: sessionId, + provider: modelEntry.provider, + model_id: modelEntry.model_id, + model_context_window: modelEntry.context_window ?? "unknown", + system_prompt: meta.system_prompt, + tools: rt.tools, + thinking_level: this.state.systemConfig.model?.thinking_level ?? "default", + agent_state: this.state.stateDir, + workspace: workspaceDir, + }, + llm: rt.llm, + environment: rt.environment, + trace, + createLLM: rt.createLLM, + createBareLLM: rt.createBareLLM, + compaction: rt.compaction, + // Model doesn't support images: input images are written to the session scratchpad and their paths appended to the text (viewed via describe_image). + ...(modelEntry.vision === false + ? { + inputImagesDir: path.join( + scratchpadDir(this.state.root, this.state.projectId, this.state.agentId), + sessionId, + ), + } + : {}), + ...(this.state.systemConfig.max_turns !== undefined + ? { maxTurns: this.state.systemConfig.max_turns } + : {}), + // session_meta is already in the original Trace file, so it isn't rewritten; on the first write after a compaction-triggered rotation, the file is split first. + metaAlreadyWritten: true, + initialEngineState: { + carryOver: resumed.carryOver, + ...(resumed.pendingSummary ? { pendingSummary: resumed.pendingSummary } : {}), + sessionTurns: resumed.sessionTurns, + sessionTokens: resumed.sessionTokens, + lastRequestTotal: resumed.lastRequestTotal, + pendingTraceRotation: resumed.contextClosed, + }, + resumedHistory: resumed.renderMessages, + }); + } + + /** Id of the most recent Session under the current Agent (determined by the timestamp in session_id); returns null if there is no Session. */ + async latestSessionId(): Promise { + return latestTraceSessionId( + tracesDir(this.state.root, this.state.projectId, this.state.agentId), + ); + } + + /** + * Assemble a Session's runtime components (shared by createSession and + * resumeSession): the child-Agent runner, Environment and tools, the LLM object + * and its post-compaction rebuild factory, and the compaction config. + */ + private async buildRuntime(args: { + workspaceDir: string; + /** This Session's Model entry: the caller (createSession / resumeSession) has already validated it exists in the config. */ + modelEntry: ModelEntry; + apiKey: string | undefined; + baseUrl: string | undefined; + systemPrompt: string; + subagentDepth: number; + vault: Record; + }): Promise<{ + environment: Environment; + tools: ToolDefinition[]; + llm: GenerativeModel; + createLLM: (sessionTokens: TokenCounts) => GenerativeModel; + createBareLLM: () => GenerativeModel; + compaction: CompactionSettings; + }> { + const { workspaceDir, modelEntry, apiKey, baseUrl, systemPrompt, subagentDepth, vault } = args; + // Child-Agent runner: injected into the run_subagent tool so it doesn't need to + // depend on Agent/Session (breaking a circular dependency). The model can + // optionally choose agentId (omitted = call the current Agent) and modelId + // (omitted = Project default). Precheck errors (depth limit exceeded / agent + // doesn't exist) are expressed as throws, which the Environment collapses to failed. + // Docs: /docs/interfaces § "Subagent interfaces" + const parentAgent = this; + const { root, projectId, agentId: parentAgentId } = this.state; + const subagentRunner: SubagentRunner = { + // Spawn and run are separate: the same child Session can run for multiple turns + // (continuing via input_subagent appending a prompt); resource cleanup is + // consolidated in handle.dispose (called by the managing ManagedSubagentSession). + async spawn({ agentId, modelId }) { + if (subagentDepth >= MAX_SUBAGENT_DEPTH) { + throw new Error( + `subagent depth limit ${MAX_SUBAGENT_DEPTH} reached; not spawning another subagent`, + ); + } + if (agentId !== undefined && agentId !== parentAgentId) { + try { + assertValidId("agent_id", agentId); + await fs.access(systemConfigPath(root, projectId, agentId)); + } catch { + throw new Error( + `subagent error: agent "${agentId}" does not exist or is not accessible`, + ); + } + } + const childAgent = + agentId !== undefined && agentId !== parentAgentId + ? await createAgent({ root, projectId, agentId }) + : parentAgent; + const childSession = await childAgent.createSession({ + workspaceDir, + ...(modelId !== undefined ? { modelId } : {}), + subagentDepth: subagentDepth + 1, + }); + // All child-session messages are tagged with an origin (the child Session id, + // prepended as one hop from outer to inner); the first turn forwards the + // child's session_meta first (including agent_state and other metadata) so the + // parent frontend can recognize the nested session (for rendering, stats, + // approval visibility); the parent Trace skips these accordingly (the child + // Session has its own Trace, linked by session id). + const hop: MessageOrigin = childSession.sessionId; + let metaSent = false; + return { + sessionId: hop, + async *run({ prompt, signal, approve }) { + if (!metaSent) { + metaSent = true; + yield withOrigin(childSession.metaMessage, hop); + } + // Pass through the parent's approval callback: the child Session inherits + // the parent Agent's approval mode (with no callback, the child engine + // defaults to deny). The tool_call received for approval also carries the + // origin, so the approval UI can identify which tool a subagent is calling. + const childApprove = approve + ? (tc: OmniMessage) => approve(withOrigin(tc, hop)) + : undefined; + for await (const msg of childSession.run([userText(prompt)], { + ...(signal ? { signal } : {}), + ...(childApprove ? { approve: childApprove } : {}), + })) { + yield withOrigin(msg, hop); + } + }, + dispose() { + childSession.dispose(); + }, + }; + }, + }; + + // Tool exposure is capped by depth: a (leaf) child Agent that has reached the + // max spawn depth no longer gets run_subagent or input_subagent (the latter + // depends on the subagent_id produced by the former, so exposing it alone is + // meaningless). + const canSpawn = subagentDepth < MAX_SUBAGENT_DEPTH; + const baseToolConfig = buildToolConfig(this.state); + // Select tool entries by the session model's type (marked via forModel: vision + // models use read_image, text-only models use describe_image; entries without + // this marker are unaffected). + const modelVision = modelEntry.vision !== false; + let customTools = selectBuiltinToolsForModel(baseToolConfig.customTools, modelVision); + if (!canSpawn) { + customTools = customTools.filter( + (d) => d.name !== SUBAGENT_NAME && d.name !== INPUT_SUBAGENT_NAME, + ); + } + const toolConfig = { ...baseToolConfig, customTools }; + + // When the session model doesn't support images (vision=false): inject a vision + // model service for describe_image (forModel: "text-only", selected by the filter + // above) — images are described by the Project config's vision_model (a paired + // reference), and the tool returns text. Even when unconfigured or invalid, it is + // still injected (modelId=null); the tool then finishes with a failed explanation, + // and images are never allowed into that session's history. + let visionDescriber: VisionDescriberService | undefined; + if (modelEntry.vision === false) { + const visionRef = this.projectConfig.vision_model; + const visionEntry = visionRef ? getModel(this.projectConfig, visionRef) : undefined; + if (visionEntry && visionEntry.vision !== false) { + visionDescriber = { + // The model attribution in the tool output matches the request's source: both are the entry's upstream model_id. + modelId: visionEntry.model_id, + createLLM: () => + new GenerativeModel({ + modelId: visionEntry.model_id, + ...(visionEntry.api_key !== undefined ? { apiKey: visionEntry.api_key } : {}), + ...(visionEntry.base_url !== undefined ? { baseUrl: visionEntry.base_url } : {}), + ...(visionEntry.client_type !== undefined + ? { clientType: visionEntry.client_type } + : {}), + tools: [], + thinkingLevel: "none", + maxTokens: 2048, + requestTimeoutMs: 60_000, + }), + }; + } else { + visionDescriber = { modelId: null }; + } + } + + // Environment binds the Workspace and tool config; tools are listed first so + // GenerativeModel can be initialized. Vault environment variables are injected + // into command subprocesses (shared by createSession and resumeSession; the + // caller reads the current agent_state/.vault.toml); a child Agent loads **its + // own** vault via createAgent rather than inheriting the parent's. + const environment = new Environment({ + workspaceDir, + toolConfig, + services: { subagentRunner, ...(visionDescriber ? { visionDescriber } : {}) }, + ...(Object.keys(vault).length > 0 ? { vault } : {}), + }); + const tools = await environment.listTools(); + + // LLM constructor args are extracted into a constant so they can be reused as-is when + // rebuilding a new LLM object after compaction (with a fresh model context) — the system + // prompt and tool definitions aren't part of the compacted history, so the new object keeps + // them unchanged. The model id sent to AgentHub is always the entry's upstream `model_id` + // (client_type inference/passing follows it); session_meta, Trace, usage, pricing, and catalog + // matching all use the (provider, model_id) pair as the primary key. + // The tool_call_id uniqueness registry is shared with the new LLM rebuilt from llmConfig after + // compaction: its uniqueness scope is the Session's whole render span, so same-named tool calls + // after compaction don't collide with earlier tool cards' ids. + const llmConfig: GenerativeModelConfig = { + modelId: modelEntry.model_id, + toolCallIds: new ToolCallIdAllocator(), + ...(apiKey !== undefined ? { apiKey } : {}), + ...(baseUrl !== undefined ? { baseUrl } : {}), + ...(modelEntry.client_type !== undefined ? { clientType: modelEntry.client_type } : {}), + tools, + systemPrompt, + ...(modelEntry.context_window !== undefined + ? { contextWindow: modelEntry.context_window } + : {}), + ...(this.state.systemConfig.model?.max_tokens !== undefined + ? { maxTokens: this.state.systemConfig.model.max_tokens } + : {}), + ...(this.state.systemConfig.model?.thinking_level !== undefined + ? { thinkingLevel: this.state.systemConfig.model.thinking_level } + : {}), + ...(this.state.systemConfig.model?.timeoutMs !== undefined + ? { requestTimeoutMs: this.state.systemConfig.model.timeoutMs } + : {}), + }; + const llm = new GenerativeModel(llmConfig); + const createLLM = (sessionTokens: TokenCounts): GenerativeModel => { + const next = new GenerativeModel(llmConfig); + // Carries over the Session's cumulative Token counts, so token_usage.session stays continuous across compaction. + next.sessionTokens = sessionTokens; + return next; + }; + // Bare LLM for one-off out-of-band requests (meta requests like generateTitle): + // same Model/credentials, no tools, no system prompt, thinking disabled, a small + // output cap, and an independent timeout. + const createBareLLM = (): GenerativeModel => + new GenerativeModel({ + modelId: modelEntry.model_id, + ...(apiKey !== undefined ? { apiKey } : {}), + ...(baseUrl !== undefined ? { baseUrl } : {}), + ...(modelEntry.client_type !== undefined ? { clientType: modelEntry.client_type } : {}), + tools: [], + thinkingLevel: "none", + maxTokens: 300, + requestTimeoutMs: 30_000, + }); + + // Compaction config: defaults are filled in here; an unknown mode falls back to summarize (the default). + const compactionConfig = this.state.systemConfig.compaction; + const compaction: CompactionSettings = { + maxContextLength: effectiveMaxContextLength( + compactionConfig?.max_context_length ?? 128000, + modelEntry.context_window, + ), + maxSessionTurns: compactionConfig?.max_session_turns ?? -1, + mode: compactionConfig?.mode === "discard" ? "discard" : "summarize", + prompt: compactionConfig?.prompt ?? DEFAULT_COMPACTION_PROMPT, + }; + + return { environment, tools, llm, createLLM, createBareLLM, compaction }; + } +} diff --git a/packages/core/src/engine/context-engine.ts b/packages/core/src/engine/context-engine.ts new file mode 100644 index 0000000..6cef147 --- /dev/null +++ b/packages/core/src/engine/context-engine.ts @@ -0,0 +1,1152 @@ +/** + * context_engine — orchestrates the ReAct loop. + * + * context_engine only handles OmniMessage, orchestrating the flow of events between the + * Human, LLM, and Environment interfaces, and writes every observable action to Trace. + * The initial version keeps a linear message history. + * + * Human is the SDK's input/output boundary: there is no "Human implementation/interface". + * Input is the Prompt list passed to `run`, plus the abort signal `signal` and the + * per-tool approval callback `approve` in `RunOptions`; output is the OmniMessage stream + * produced by `run`. + * + * Docs: packages/docs/content/agent-loop.{zh,en}.md (site path /docs/agent-loop) documents + * the turn lifecycle, carry-over, reconnect and compaction implemented here. + * + * Approval is an **in-turn interaction** and tool calls are **async/incremental** (see + * comment #24): + * - A single `run` call automatically runs the entire ReAct loop (no more resuming in batches); + * - Each tool_call is emitted as soon as its stream completes → `await approve` → if + * allowed, it runs via Environment; + * - Execution **does not block** continued consumption of the LLM stream or approval of + * the next tool (executions can overlap), but approvals still happen one at a time; + * - partial/complete `tool_call_output` is yielded in **completion order**; + * - once all tool outputs for the turn are ready, they become the next turn's LLM input; + * the Task ends once a turn produces no more tool_call. + * + * Implementation note: an internal queue merges "the LLM event stream + N concurrent tool + * output streams" into a single yield sequence. GenerativeModel is a stateful object + * (AgentHub maintains the history); each turn the engine only hands it the "new" messages: + * the user Prompt on the first turn, and the previous turn's tool_call_output afterward. + */ +import { + abortEvent, + approvalDecision, + assistantText, + compactionBegin, + compactionEnd, + emptyTokenCounts, + isCompleteModelMessage, + isSessionMeta, + partialText, + requestBegin, + requestEnd, + subagentEvent, + toolCallOutput, + userText, +} from "../omnimessage/index.js"; +import type { + ApprovalDecision, + CompactionMode, + CompactionReason, + OmniMessage, + StopReason, + TextPayload, + ThinkingPayload, + TokenCounts, + TokenUsagePayload, + ToolCallOutputPayload, + ToolCallPayload, +} from "../omnimessage/index.js"; +import type { ApproveFn, EnvironmentInterface, LLMInterface, LLMOutcome } from "../interfaces.js"; + +/** Trace sink: `write` a complete/event/meta message; `rotate` starts a new file (compaction splits files). */ +export interface TraceSink { + write(msg: OmniMessage): Promise; + /** Optional: start a new Trace file (index+1), used to record the new model context after compaction. */ + rotate?(): Promise; +} + +/** + * Resolved context compaction settings (defaults filled in by the composition layer). + * Docs: /docs/agent-loop § "Compaction". + */ +export interface CompactionSettings { + /** Context token threshold (uses the most recent token_usage's request.total); <=0 disables it. */ + maxContextLength: number; + /** Session cumulative turn threshold (counted per LLM Request, across Tasks); <=0 means no limit. */ + maxSessionTurns: number; + mode: CompactionMode; + /** Prompt used for summarize compaction. */ + prompt: string; +} + +/** Result of one compaction run: status is a terminal state (completed / failed / aborted); carries the summary message when summarize succeeds. */ +interface CompactionResult { + status: StopReason; + summary?: OmniMessage; +} + +/** + * Options for `run`. + * Docs: /docs/agent-loop § "Inputs and outputs". + */ +export interface RunOptions { + /** Abort signal (e.g. Ctrl-C). */ + signal?: AbortSignal; + /** Per-tool approval callback; defaults to denying everything (conservative, to avoid accidental approval when unattended). */ + approve?: ApproveFn; +} + +/** + * Engine initial state (used for Session resumption): derived by replaying Trace, so the + * resumed engine behaves the same as before the process + * exited. Not passed when creating a normal new Session. + */ +export interface EngineInitialState { + /** Pending input (carry-over): resent alongside new input on the first `run` after resumption (synthetic placeholders exist only in memory, never written to Trace). */ + carryOver?: OmniMessage[]; + /** Summary recovered from a completed summarize compaction: used as the prefix of the next `run` input (merged with the user Prompt). */ + pendingSummary?: OmniMessage; + /** Carried-over Session cumulative turn count. */ + sessionTurns?: number; + /** Carried-over Session cumulative token counts (handed to the new object when compaction swaps it in). */ + sessionTokens?: TokenCounts; + /** Most recent token_usage's request.total (the context usage figure, keeps compaction threshold checks continuous). */ + lastRequestTotal?: number; + /** Recovered from a completed compaction: the context is already closed, so rotate the Trace file (index+1, writing session_meta) before the first write. */ + pendingTraceRotation?: boolean; +} + +export interface ContextEngineDeps { + llm: LLMInterface; + environment: EnvironmentInterface; + /** Optional Trace writer; the writer is responsible for filtering out streaming partial_* messages. */ + trace?: TraceSink; + /** Engine initial state (derived by replaying Trace on Session resumption). */ + initialState?: EngineInitialState; + /** Maximum LLM turns for a single Task. Defaults to 100. */ + maxTurns?: number; + /** Maximum automatic retries for LLM timeout/reconnect within a single run. Defaults to 2. */ + maxReconnects?: number; + /** Linear backoff base (ms) before each reconnect retry; actual backoff = base × retry number. Defaults to 250. */ + reconnectBackoffMs?: number; + /** + * Creates a new LLM object after compaction (a fresh model context); the argument is the + * current Session cumulative token counts, for the new object to carry forward + * (token_usage.session stays continuous across compaction). Context compaction is + * unavailable if this is not provided. + */ + createLLM?: (sessionTokens: TokenCounts) => LLMInterface; + /** Context compaction settings; only takes effect if provided together with `createLLM`. */ + compaction?: CompactionSettings; + /** This Session's session_meta message; written at the start of the new Trace file after compaction splits it. */ + sessionMeta?: OmniMessage; +} + +/** Whether compaction is possible; when not `ok`, `compact()` is a no-op and yields no messages (see ContextEngine.compactability). */ +export type CompactAvailability = "ok" | "unsupported" | "empty" | "just_compacted"; + +/** Result of executing one LLM turn (the return value of runTurn). */ +interface TurnResult { + /** All tool outputs for this turn, reordered to match the original tool_call order (for the next turn's LLM input). */ + toolOutputs: OmniMessage[]; + /** tool_calls issued by the model this turn (in original order, real requests only). */ + toolCalls: OmniMessage[]; + /** Complete thinking/text segments produced by the model this turn (including partial segments finalized on interruption), for carry-over flattening. */ + assistantSegments: OmniMessage[]; + /** Terminal state of this turn's LLM request (completed / failed / aborted / timeout / malformed). */ + outcome: LLMOutcome; +} + +/** + * Merge queue: lets multiple concurrent producers (the LLM stream consumer + several tool + * executions) push OmniMessage entries; a single consumer (run's generator) pulls and + * yields them in push order. Finishes once all producers are done and the queue is drained. + * Docs: /docs/message-flow § "The merge point: MergeQueue". + */ +class MergeQueue { + private items: OmniMessage[] = []; + private producers = 0; + private wake: (() => void) | null = null; + + /** Registers a producer. */ + addProducer(): void { + this.producers += 1; + } + + /** Deregisters a producer (its stream has finished). */ + removeProducer(): void { + this.producers -= 1; + this.signal(); + } + + /** Pushes a message and wakes the consumer. */ + push(msg: OmniMessage): void { + this.items.push(msg); + this.signal(); + } + + private signal(): void { + if (this.wake) { + const w = this.wake; + this.wake = null; + w(); + } + } + + /** Takes the next message; waits if empty but producers remain; returns null if empty and no producers remain. */ + async next(): Promise { + for (;;) { + if (this.items.length > 0) return this.items.shift()!; + if (this.producers === 0) return null; + await new Promise((resolve) => { + this.wake = resolve; + }); + } + } +} + +export class ContextEngine { + private readonly maxTurns: number; + private readonly maxReconnects: number; + private readonly reconnectBackoffMs: number; + /** Interruption cleanup: content to resend generated when the previous run was aborted, held on the engine across runs. */ + private pendingCarryOver: OmniMessage[] = []; + /** Current LLM object; swapped for a new one created by `createLLM` after a successful compaction (a fresh model context). */ + private llm: LLMInterface; + /** Session cumulative turn count: counted per LLM Request that produces token_usage, across Tasks; reset to zero after compaction completes. */ + private sessionTurns = 0; + /** Whether the current context was produced by a compaction (`startNewContext`); this flag becomes meaningless once a new completed turn occurs. */ + private fromCompaction = false; + /** Most recent token_usage's request.total, i.e. the current context usage figure. */ + private lastRequestTotal = 0; + /** Most recent token_usage's session cumulative counts, handed to the new LLM object when compaction swaps it in. */ + private lastSessionTokens: TokenCounts = emptyTokenCounts(); + /** Summary produced by a Task-boundary compaction: used as the prefix of the next `run` input (merged with the next user Prompt). */ + private pendingSummary: OmniMessage | null = null; + /** + * Set to true once compaction completes: Trace rotation is deferred until the next + * message that needs writing (see `write`) — so that if no further messages follow the + * compaction, we don't create an empty file containing only session_meta. + */ + private pendingTraceRotation = false; + + constructor(private readonly deps: ContextEngineDeps) { + this.maxTurns = deps.maxTurns ?? 100; + this.maxReconnects = deps.maxReconnects ?? 2; + this.reconnectBackoffMs = deps.reconnectBackoffMs ?? 250; + this.llm = deps.llm; + // Session resumption: apply the initial state derived from replay. + const init = deps.initialState; + if (init) { + this.pendingCarryOver = init.carryOver ?? []; + this.pendingSummary = init.pendingSummary ?? null; + this.sessionTurns = init.sessionTurns ?? 0; + this.lastSessionTokens = init.sessionTokens ?? emptyTokenCounts(); + this.lastRequestTotal = init.lastRequestTotal ?? 0; + this.pendingTraceRotation = init.pendingTraceRotation ?? false; + } + } + + /** + * Runs a Task to completion, streaming out OmniMessage. `newMessages` is this call's + * Prompt (only the newly added input, not the full history — history is maintained by the + * stateful GenerativeModel); `opts.signal` is the abort signal, `opts.approve` is the + * per-tool approval callback. + * Docs: /docs/agent-loop § "The loop at a glance". + */ + async *run(newMessages: OmniMessage[], opts?: RunOptions): AsyncGenerator { + const signal = opts?.signal; + // Default approval policy: deny (conservative). CLI/Web will inject a real callback (interactive or permission-mode based). + const approve: ApproveFn = opts?.approve ?? (async () => "deny"); + + // Merge the Task-boundary compaction summary (the new context's first input, merged with + // this Prompt), the carry-over left over from the last interruption, and this call's new + // input, to form this Request's input. + const summary = this.pendingSummary; + this.pendingSummary = null; + const carryOver = this.pendingCarryOver; + this.pendingCarryOver = []; + const prefix = summary ? [summary, ...carryOver] : carryOver; + const input = prefix.length ? [...prefix, ...newMessages] : newMessages; + + // Input is written to Trace (Prompt record, incl. audit trail) but not replayed to + // the render layer. carry-over is not written to Trace: real messages (tool outputs etc.) + // are already written when produced; synthetic content (flatten text, backfilled + // placeholders) is **sent to the model only, never persisted** — Trace records only real + // messages, and resumption replay best-effort reconstructs from original messages. + // Exception: the compaction summary, which is the new + // context's first input record, is written as usual. + if (summary) await this.write(summary); + for (const msg of newMessages) await this.write(msg); + + if (signal?.aborted) { + // Aborted before the Request was issued: the input is held **as-is** as carry-over + // (trailing-input semantics: input the Request never got to send is kept unchanged) + // — not flattened, so replay matches in-process behavior and + // multimodal input isn't lost. The message is already written to Trace, so it won't be + // rewritten on the next send. + this.pendingCarryOver = input; + yield* this.emitAbort("aborted by user"); + return; + } + + let turnCount = 0; + // Each turn's LLM input: the first turn is the Prompt, later turns are the previous turn's + // tool outputs. + let nextInput: OmniMessage[] = input; + + for (;;) { + // max_turns guard: emit a length notice and stop once exceeded. + if (turnCount >= this.maxTurns) { + // This turn's pending input (usually the previous turn's tool outputs) was never + // submitted to the LLM: hold it as carry-over, to be resent merged with new input on + // the next `run` (same as interruption-cleanup case A) — the previous turn's assistant + // tool_call has already been committed by AgentHub, so discarding its paired output and + // sending a fresh message would be rejected by the provider as an unanswered tool_use + // (400, see issue #33). + this.pendingCarryOver = nextInput; + yield* this.emitMaxTurns(); + return; + } + turnCount += 1; + + // This turn's input. A timeout/malformed attempt is never committed to history by + // AgentHub (an abnormally interrupted stream doesn't land in history), so reconnect + // resends this turn's input unchanged, appending a `` block carrying what + // the failed attempt already produced — the model continues from there instead of + // re-running tools; the tag is distinct from the user-interruption ``. + const failedTurns: TurnResult[] = []; + let attemptInput = nextInput; + let reconnects = 0; + let turn: TurnResult; + + for (;;) { + // Both LLM and Environment handle errors internally and guarantee a complete, closed + // output with no thrown exceptions; the engine doesn't handle exceptions — + // it decides retry/resend purely from `outcome`. + turn = yield* this.runTurn(attemptInput, approve, signal); + + // User interruption (the LLM stream was aborted, outcome=aborted, or `signal` fired + // during tool execution): stop and hand control back to the user. + if (signal?.aborted || turn.outcome.status === "aborted") { + this.pendingCarryOver = this.buildCarryOver(attemptInput, turn); + yield* this.emitAbort("aborted by user"); + return; + } + // Non-retryable error (auth/parameter etc.): stop and hand control back to the user; + // the failure reason is written to the abort event / Trace. + if (turn.outcome.status === "failed") { + this.pendingCarryOver = this.buildCarryOver(attemptInput, turn); + yield* this.emitAbort(`llm request error: ${turn.outcome.message ?? "unknown"}`); + return; + } + // Completed normally. + if (turn.outcome.status === "completed") break; + + // Only timeout / malformed remain: reconnect automatically within the same run. When + // retries are exhausted or the backoff is interrupted, the retry input is held as-is as + // carry-over (the original input is already written to Trace, so it isn't rewritten). + // The frontend surfaces the retry process and count via request_end(timeout|malformed) + // followed by the next request_begin. + failedTurns.push(turn); + attemptInput = this.withRetriedTurns(nextInput, failedTurns); + if (reconnects >= this.maxReconnects) { + this.pendingCarryOver = attemptInput; + const reason = turn.outcome.status === "malformed" ? "malformed response" : "reconnect"; + yield* this.emitAbort(`${reason} failed after ${this.maxReconnects} retries`); + return; + } + reconnects += 1; + if (!(await this.backoff(reconnects, signal))) { + this.pendingCarryOver = attemptInput; + yield* this.emitAbort("aborted during reconnect backoff"); + return; + } + } + + // Compaction checkpoint: after every LLM Request produces token_usage. This also + // applies mid-Task — when runTurn returns, all of this turn's + // tool results are ready and paired with their tool_call. + const midTask = turn.toolOutputs.length > 0; + const compactionReason = this.compactionTrigger(); + if (compactionReason) { + const mode = this.deps.compaction!.mode; + if (mode === "discard") { + // Once discarded, the current Task can't continue: if mid-Task, defer until this + // Task ends. + if (!midTask) { + yield* this.discardContext(compactionReason); + return; + } + } else { + const result = yield* this.summarizeContext( + compactionReason, + midTask ? turn.toolOutputs : [], + signal, + ); + if (result.status === "aborted") { + // User interrupted compaction: keep the original context; if mid-Task, hold the + // tool outputs as carry-over per case A. + if (midTask) { + this.pendingCarryOver = this.buildCarryOver(attemptInput, turn); + yield* this.emitAbort("aborted during compaction"); + } + return; + } + if (result.status === "completed") { + if (!midTask) { + // Task boundary: the summary is merged with the next user Prompt as the new + // context's first input. + this.pendingSummary = result.summary!; + return; + } + // Mid-Task: the summary itself becomes the new LLM object's first input (this + // turn's tool results were already folded into the compaction request and absorbed + // into the summary); continuation relies on the model's own next-step plan written + // into the summary, with no hardcoded continuation instruction appended. + await this.write(result.summary!); + nextInput = [result.summary!]; + continue; + } + // failed: keep the original context and Trace index; the current Task continues and + // retries on the next trigger (no fallback to discard). + } + } + + // No tool_call this turn -> the Task ends (the final reply has already been streamed out). + if (!midTask) return; + // Otherwise continue, using the tool outputs as the next turn's LLM input. + nextInput = turn.toolOutputs; + } + } + + /** + * User-initiated compaction request (e.g. a CLI command): reuses the automatic compaction + * flow without checking thresholds (reason=manual). Only callable at a Task boundary (between + * runs); streams out paired compaction events. No-op when compaction is not configured. + * + * Carry-over left over from an interruption is cleaned up here too: summarize folds it into + * the compaction request (structured tool outputs keep their pairing with the already + * committed tool_call, otherwise the compaction request itself would be rejected by the + * provider as an unanswered tool_use, see issue #33; flatten text is absorbed into the + * summary); discard drops the structured outputs paired with the old context, keeping only the + * self-contained flatten text. + */ + /** + * Whether compaction is possible, and the **reason** when it isn't. + * + * `compact()` is a no-op and **yields no messages** in these cases; if the UI treats invoking + * it as a successful start, it ends up waiting forever for a compaction banner that never + * arrives — that's exactly how "/compact does nothing after an interruption" happens. Callers + * (Web / CLI) should give feedback upfront based on this. + * + * - `unsupported`: compaction capability is not configured; + * - `empty`: the current context hasn't completed a single turn (`sessionTurns` only + * increments when `token_usage` arrives — a turn only counts once the request finishes + * normally, so it's still 0 right after the first request is interrupted); + * - `just_compacted`: no new conversation since the last compaction. Both cases have + * `sessionTurns` === 0, but they mean two completely different things to the user and must + * not be conflated. + */ + compactability(): CompactAvailability { + if (!this.deps.compaction || !this.deps.createLLM) return "unsupported"; + if (this.sessionTurns > 0) return "ok"; + return this.fromCompaction ? "just_compacted" : "empty"; + } + + async *compact(opts?: { signal?: AbortSignal }): AsyncGenerator { + if (!this.deps.compaction || !this.deps.createLLM) return; + // The current context has no completed LLM turns: nothing to compact, return immediately. + // This also guards against two /compact calls in a row — the new context is empty right + // after the previous compaction, so running again would overwrite the not-yet-consumed + // pendingSummary with an "empty summary," permanently losing the only record of the prior + // conversation. + if (this.sessionTurns === 0) return; + if (this.deps.compaction.mode === "discard") { + this.pendingCarryOver = this.pendingCarryOver.filter( + (m) => (m.payload as { type?: string }).type !== "tool_call_output", + ); + yield* this.discardContext("manual"); + return; + } + const result = yield* this.summarizeContext("manual", this.pendingCarryOver, opts?.signal); + if (result.status === "completed") { + this.pendingCarryOver = []; + this.pendingSummary = result.summary!; + } + } + + /** + * Linear backoff before a reconnect retry (base × retry number, numbering starts at 1); + * returns false if the user interrupts during the backoff, so the caller can proceed to + * interruption cleanup. + */ + private backoff(attempt: number, signal?: AbortSignal): Promise { + const ms = this.reconnectBackoffMs * attempt; + return new Promise((resolve) => { + if (signal?.aborted) { + resolve(false); + return; + } + const onAbort = (): void => { + clearTimeout(timer); + resolve(false); + }; + const timer = setTimeout(() => { + signal?.removeEventListener("abort", onAbort); + resolve(true); + }, ms); + signal?.addEventListener("abort", onAbort, { once: true }); + }); + } + + /** + * Runs one LLM turn: consumes the LLM stream, approving each complete tool_call immediately; + * "allow" runs it concurrently (without blocking further stream consumption/approval), "deny" + * feeds back an aborted output. partial/complete tool_call_output is yielded in completion + * order. Returns all of this turn's tool outputs (for the next turn) and whether it was + * interrupted midway. + * Docs: /docs/agent-loop § "Lifecycle of a turn". + */ + private async *runTurn( + input: OmniMessage[], + approve: ApproveFn, + signal?: AbortSignal, + ): AsyncGenerator { + const queue = new MergeQueue(); + // Tool outputs are collected in **completion order** (for streaming yield to the frontend); + // the tool_calls' **original order** is recorded separately, and reordered back to original + // order when fed into the next LLM turn (async tool calls: feedback order is preserved). + const toolOutputs: OmniMessage[] = []; + const toolCalls: OmniMessage[] = []; + const callOrder: string[] = []; + // This turn's complete thinking/text segments produced by the model (including partial + // segments finalized on interruption), for carry-over flatten. + const assistantSegments: OmniMessage[] = []; + // This turn's LLM terminal state: taken from streamGenerate's generator return value. + let outcome: LLMOutcome = { status: "completed" }; + + // Driver task: consumes the LLM stream + approves one at a time + dispatches tool + // execution. It is itself a producer. + queue.addProducer(); + const drive = (async () => { + try { + // Request boundary events (replayability): start is + // emitted when the request is issued, stop carries the terminal state at completion — + // replay mechanically determines from these whether the turn was committed by AgentHub. + const startEvt = requestBegin(); + queue.push(startEvt); + await this.write(startEvt); + // Iterate manually to capture the generator's **return value** (LLMOutcome); LLM + // guarantees it never throws. + const gen = this.llm.streamGenerate({ + newMessages: input, + ...(signal ? { signal } : {}), + }); + for (;;) { + const res = await gen.next(); + if (res.done) { + outcome = res.value; + const stopEvt = requestEnd(outcome.status); + queue.push(stopEvt); + await this.write(stopEvt); + break; + } + const msg = res.value; + queue.push(msg); + await this.write(msg); + // token_usage means "this Request completed normally": record the context usage / + // Session cumulative counts, and increment the Session turn count (counted per LLM + // Request, across Tasks; used for compaction threshold checks). + if (this.observeTokenUsage(msg)) this.sessionTurns += 1; + // Collect complete thinking/text segments (including partial segments finalized on + // interruption), for carry-over flatten. + if ( + isCompleteModelMessage(msg) && + (msg.payload.type === "thinking" || msg.payload.type === "text") + ) { + assistantSegments.push(msg); + } + // Approve as soon as each real, complete tool_call finishes streaming. A tool_call + // synthesized to close out an interruption carries a non-"completed" stop_reason (see + // finishInterrupted): its arguments weren't fully emitted, and it exists only + // for structural closure and observability — it isn't dispatched for execution, isn't + // added to this turn's ledger, and gets no paired output backfilled: such a tool_call + // was never committed to history by AgentHub, so there's nothing to pair. This turn + // must then end with a non-completed outcome (only interruption closure produces such + // a tool_call): timeout/malformed is cleaned up by reconnect resending the flatten + // carry-over, failed/aborted exits directly. + if (isCompleteModelMessage(msg) && msg.payload.type === "tool_call") { + const tc = msg as OmniMessage; + if (tc.payload.stop_reason !== "completed") continue; + const toolCallId = tc.payload.tool_call_id; + callOrder.push(toolCallId); + toolCalls.push(tc); + // Already interrupted: stop dispatching new tools, but keep consuming until the LLM + // returns its outcome (the LLM will close out quickly and return aborted). + if (signal?.aborted) continue; + // The approval callback is injected externally (RunOptions.approve): any throw + // collapses to deny (conservative), so the exception never escapes the engine — + // otherwise it would propagate through session.run without building carry-over, + // leaving the already-committed tool_use unanswered. + let decision: ApprovalDecision; + try { + decision = await approve(tc); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + process.stderr.write(`[penguin] approve callback threw: ${message}; denying.\n`); + decision = "deny"; + } + if (signal?.aborted) continue; + // approve is a callback; context_engine emits its decision as an approval_decision + // OmniMessage: pushed to the stream for frontend rendering, and written to Trace. + const decisionMsg = approvalDecision(decision, toolCallId); + queue.push(decisionMsg); + await this.write(decisionMsg); + if (decision !== "allow") { + // User denied: feed back an aborted output, indicating the tool call was + // manually canceled. + const denied = toolCallOutput({ + output: "Tool call denied by user.", + toolCallId, + stopReason: "aborted", + }); + queue.push(denied); + await this.write(denied); + toolOutputs.push(denied); + continue; + } + // Approved: run concurrently, without blocking further consumption of the LLM + // stream or approval of the next tool. + queue.addProducer(); + void this.executeOne(tc, queue, toolOutputs, signal, approve).finally(() => { + queue.removeProducer(); + }); + } + } + } finally { + queue.removeProducer(); + } + })(); + + // Single consumer: yield merged messages one at a time until all producers are done and + // the queue is drained. + for (;;) { + const msg = await queue.next(); + if (msg === null) break; + yield msg; + } + // Wait for the driver task to fully finish (state settles). + await drive; + + // Feed into the next turn: reordered to the original tool_call order (each tool_call has + // exactly one output, see the executeOne invariant). + const byId = new Map(); + for (const out of toolOutputs) { + const id = (out.payload as { tool_call_id?: string }).tool_call_id; + if (id !== undefined) byId.set(id, out); + } + const orderedOutputs: OmniMessage[] = []; + const seen = new Set(); + for (const id of callOrder) { + if (seen.has(id)) continue; // Dedupe: feed back exactly one output per tool_call_id, to preserve pairing + seen.add(id); + const out = byId.get(id); + if (out) orderedOutputs.push(out); + } + return { toolOutputs: orderedOutputs, toolCalls, assistantSegments, outcome }; + } + + /** + * Executes a single approved tool: streams its partial/complete tool_call_output (through the + * queue), and collects the complete tool_call_output into toolOutputs. + * + * Environment is contracted to handle all errors internally: it guarantees exactly one + * complete `tool_call_output` to close out and never throws. But since + * EnvironmentInterface can be injected by consumers via a public API, if a contract-violating + * exception escapes, this fire-and-forget promise would take down the process with an + * unhandled rejection, and the missing output would leave the already-committed tool_use + * unanswered (the next request gets rejected by the provider) — so a boundary safety net is + * kept here, collapsing a contract-violating exception into a failed output. This guarantees + * exactly one complete output per tool enters toolOutputs, keeping tool_use and tool_result + * paired. + */ + private async executeOne( + toolCall: OmniMessage, + queue: MergeQueue, + toolOutputs: OmniMessage[], + signal?: AbortSignal, + approve?: ApproveFn, + ): Promise { + let completed = false; + try { + for await (const out of this.deps.environment.executeTool({ + toolCall, + ...(signal ? { signal } : {}), + // Pass through the parent approval callback: run_subagent uses this so the child + // Session inherits the parent Agent's approval mode. + ...(approve ? { approve } : {}), + })) { + queue.push(out); + // Nested-session messages carrying an origin: forwarded to the frontend as a stream; + // their content is not written to the parent Trace (the child Session has its own + // Trace). When a direct child session's (origin length 1) session_meta arrives, write a + // subagent pointer event to the parent Trace (recording only the child Session id), so + // reopening the session can recursively expand child Traces — pointers for grandchild + // sessions are recorded by the child Trace itself, so only depth 1 is recognized here. + // Never fed back — a child session's tool_call_output has no pairing with the parent's + // tool_call, and feeding it back by mistake would be rejected by the Provider. + if (out.origin && out.origin.length > 0) { + if (isSessionMeta(out) && out.origin.length === 1) { + await this.write(subagentEvent(out.origin[0]!)); + } + continue; + } + await this.write(out); + if (isCompleteModelMessage(out) && out.payload.type === "tool_call_output") { + toolOutputs.push(out); + completed = true; + } + } + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + if (completed) { + // Thrown only after the complete output was ready: pairing is intact, so just warn. + process.stderr.write(`[penguin] environment threw after tool output: ${message}\n`); + return; + } + const failed = toolCallOutput({ + output: `[tool error] ${message}`, + toolCallId: toolCall.payload.tool_call_id, + stopReason: "failed", + }); + queue.push(failed); + await this.write(failed); + toolOutputs.push(failed); + } + } + + /** Max turns reached: emits a failed notice (streaming fragments + complete text) for CLI/frontend rendering. */ + private async *emitMaxTurns(): AsyncGenerator { + // Reduce leading newlines: avoid stacking extra newlines before the text (comment #15). + const text = `[reached max turns (${this.maxTurns}); stopping]`; + const partials = [ + partialText("start"), + partialText("delta", text), + partialText("stop", "", "failed"), + ]; + for (const partial of partials) { + yield partial; + await this.write(partial); + } + const note = assistantText(text, "failed"); + yield note; + await this.write(note); + } + + // ------------------------------------------------------------------------- + // Context compaction + // ------------------------------------------------------------------------- + + /** + * Checks the compaction threshold: triggers once context usage (the most recent + * token_usage's request.total) or the Session cumulative turn count **reaches** the threshold + * (>=) — e.g. maxSessionTurns=1 compacts as soon as turn 1 completes, without waiting for the + * next Task; when both are configured, either reaching its threshold triggers compaction. + * Never triggers when compaction is not configured. + * Docs: /docs/agent-loop § "Compaction". + */ + private compactionTrigger(): CompactionReason | null { + const settings = this.deps.compaction; + if (!settings || !this.deps.createLLM) return null; + if (settings.maxContextLength > 0 && this.lastRequestTotal >= settings.maxContextLength) { + return "context"; + } + if (settings.maxSessionTurns > 0 && this.sessionTurns >= settings.maxSessionTurns) { + return "turns"; + } + return null; + } + + /** Records context usage and Session cumulative counts from a token_usage event; returns whether the message is a token_usage. */ + private observeTokenUsage(msg: OmniMessage): boolean { + if (msg.type !== "event_msg") return false; + const payload = msg.payload as Partial; + if (payload.type !== "token_usage") return false; + if (payload.request) this.lastRequestTotal = payload.request.total; + if (payload.session) this.lastSessionTokens = payload.session; + return true; + } + + /** + * `discard` compaction: sends no compaction request, simply discards the old context — + * swaps in a new LLM object and splits a new Trace file, with the next turn's input used + * unchanged as the new object's first input. Only runs at a Task boundary (deferred by the + * caller while mid-Task). + */ + private async *discardContext(reason: CompactionReason): AsyncGenerator { + yield* this.emitCompactionBegin(reason, "discard"); + yield* this.emitCompactionEnd(reason, "discard", "completed"); + await this.startNewContext(); + } + + /** + * `summarize` compaction: appends the compaction Prompt to the **old** LLM object (first + * folding in all of this turn's tool results when mid-Task, to keep tool_use/tool_result + * pairing), then extracts the `` and wraps it as `` user text. The + * compaction request's streamed output is not pushed to the Human output stream (it emits + * paired compaction events, plus the compaction request's `token_usage` — positioned between + * the two events, so the frontend can count compaction cost into its stats), but it is written + * to the old Trace. timeout/malformed reconnect via the existing retry mechanism, collapsing + * to failed once retries are exhausted; on failure/abort, the original context and Trace index + * are kept — it does not fall back to discard. + * Docs: /docs/agent-loop § "Compaction". + */ + private async *summarizeContext( + reason: CompactionReason, + pendingToolOutputs: OmniMessage[], + signal?: AbortSignal, + ): AsyncGenerator { + const settings = this.deps.compaction!; + yield* this.emitCompactionBegin(reason, "summarize"); + + // Compaction request input: this turn's tool results (mid-Task) or leftover interruption + // carry-over, plus the compaction Prompt. The compaction exchange is written to the old + // Trace (traceable but not pushed to the user); tool results were already written when + // executed and aren't recorded again, while carry-over's not-yet-written synthetic content + // (flatten text, backfilled placeholders) and the compaction Prompt are written now. + const prompt = userText(settings.prompt); + const input = [...pendingToolOutputs, prompt]; + await this.write(prompt); + + let reconnects = 0; + for (;;) { + if (signal?.aborted) { + yield* this.emitCompactionEnd(reason, "summarize", "aborted"); + return { status: "aborted" }; + } + const attempt = await this.runCompactionRequest(input, signal); + if (attempt.status === "completed") { + // The compaction request's token_usage is pushed to the Human output stream (already + // written to Trace in runCompactionRequest, so here it's only yielded, not rewritten); + // the frontend uses this to count compaction cost into stats and display it on the + // compaction-complete line. + if (attempt.usage) yield attempt.usage; + // Lenient extraction: if the output lacks a tag, use the entire compaction + // output as-is rather than treating it as a failure. + const summary = userText( + `\n${extractSummary(attempt.text)}\n`, + ); + yield* this.emitCompactionEnd(reason, "summarize", "completed"); + await this.startNewContext(); + return { status: "completed", summary }; + } + if (attempt.status === "aborted" || attempt.status === "failed") { + yield* this.emitCompactionEnd(reason, "summarize", attempt.status); + return { status: attempt.status }; + } + // timeout / malformed: retried via reconnect. The compaction request was never committed + // by AgentHub (case B), so the original input is resent unchanged. + if (reconnects >= this.maxReconnects) { + yield* this.emitCompactionEnd(reason, "summarize", "failed"); + return { status: "failed" }; + } + reconnects += 1; + const ok = await this.backoff(reconnects, signal); + if (!ok) { + yield* this.emitCompactionEnd(reason, "summarize", "aborted"); + return { status: "aborted" }; + } + } + } + + /** + * Issues one compaction request (an ordinary LLM Request): consumes the old LLM object's + * streamed output but **does not push it to the Human output stream** (except `token_usage` + * — captured and handed back via the return value for summarizeContext to yield); complete + * messages and events are written to the old Trace; complete text segments are collected as + * the compaction output. Token usage is counted into the Session cumulative totals (recorded + * via observeTokenUsage, for the new object to carry forward). + */ + private async runCompactionRequest( + input: OmniMessage[], + signal?: AbortSignal, + ): Promise<{ status: StopReason; text: string; usage: OmniMessage | null }> { + // The compaction request is itself an ordinary Request, emitting paired request events — + // written to the (old) Trace only, not pushed to the stream, keeping the compaction process + // invisible to Human. + await this.write(requestBegin()); + const gen = this.llm.streamGenerate({ + newMessages: input, + ...(signal ? { signal } : {}), + }); + let text = ""; + let usage: OmniMessage | null = null; + for (;;) { + const res = await gen.next(); + if (res.done) { + await this.write(requestEnd(res.value.status)); + return { status: res.value.status, text, usage }; + } + const msg = res.value; + await this.write(msg); + if (this.observeTokenUsage(msg)) usage = msg; + if (isCompleteModelMessage(msg) && msg.payload.type === "text") { + text += (msg.payload as TextPayload).text; + } + } + } + + /** + * Opens a new model context after successful compaction: swaps in a new LLM object (carrying + * forward the Session cumulative token counts), resets the Session turn count and context + * usage counter. Trace **does not** split files immediately — that's deferred until the next + * message that needs writing, when it rotates and opens with a session_meta (see `write`), + * avoiding an empty file if no further messages follow the compaction. + */ + private async startNewContext(): Promise { + this.pendingTraceRotation = true; + this.llm = this.deps.createLLM!(this.lastSessionTokens); + this.sessionTurns = 0; + this.lastRequestTotal = 0; + // Lets compactability() distinguish "just compacted" from "hasn't chatted yet" — both have + // sessionTurns === 0, but they mean two completely different things to the user (being told + // "no completed conversation turns yet" right after compacting is as good as saying nothing). + this.fromCompaction = true; + } + + /** Yields and records a compaction start event (carrying reason/mode/current context usage/Session cumulative turns). */ + private async *emitCompactionBegin( + reason: CompactionReason, + mode: CompactionMode, + ): AsyncGenerator { + const msg = compactionBegin({ + reason, + mode, + context: this.lastRequestTotal, + turns: this.sessionTurns, + }); + yield msg; + await this.write(msg); + } + + /** Yields and records a compaction stop event (carrying the result status; non-completed means compaction was abandoned). */ + private async *emitCompactionEnd( + reason: CompactionReason, + mode: CompactionMode, + status: StopReason, + ): AsyncGenerator { + const msg = compactionEnd({ reason, mode, status }); + yield msg; + await this.write(msg); + } + + /** Interruption: emits an abort event. Cleanup/resending is managed centrally by `run` via carry-over; the LLM history is never touched again. */ + private async *emitAbort(reason: string): AsyncGenerator { + const msg = abortEvent(reason); + yield msg; + await this.write(msg); + } + + /** + * Builds the interruption resend content (carry-over, interruption cleanup) + * based on the LLM's terminal state. Used only for the **exit** cleanup of + * aborted / failed (reconnect retry doesn't go through here — retry input is assembled by + * withRetriedTurns, appending `` with the failed attempt's output, distinct + * from the user-interruption ``): + * - Model output completed (case A, outcome=completed): AgentHub already committed an + * assistant turn containing `tool_call`, so it can only be resent as a structured + * `tool_call_output` to pair with it (cannot flatten, or the already-committed tool_call + * would be left unanswered and rejected). + * - Model output incomplete (case B): the `tool_call_output` in this turn's input (paired + * with the previous completed turn) is kept as-is; the text input and this turn's + * thinking/text/tool call/result are flattened into a single `` plain-text + * user message. + * Docs: /docs/agent-loop § "Interruption and carry-over". + */ + private buildCarryOver(attemptInput: OmniMessage[], turn: TurnResult): OmniMessage[] { + if (turn.outcome.status === "completed") { + // Case A: every **committed** tool_call must have a paired output. If execution was + // interrupted and some tool_calls were committed but never dispatched/completed, backfill + // an interrupted-state placeholder for each, avoiding an unanswered tool_use in the next + // turn that the provider would reject. + const haveIds = new Set( + turn.toolOutputs.map((o) => (o.payload as { tool_call_id?: string }).tool_call_id), + ); + const backfill = turn.toolCalls + .filter((tc) => !haveIds.has(tc.payload.tool_call_id)) + .map((tc) => + toolCallOutput({ + output: "[interrupted: tool aborted by user]", + toolCallId: tc.payload.tool_call_id, + stopReason: "aborted", + }), + ); + // Placeholders are sent to the model only and not written to Trace (synthetic carry-over + // isn't persisted); resumption replay re-synthesizes placeholders as needed to guarantee + // pairing (pairing fallback). Real outputs were already + // written when produced. + return backfill.length ? [...turn.toolOutputs, ...backfill] : turn.toolOutputs; + } + return this.flattenCarryOver( + attemptInput, + turn.assistantSegments, + turn.toolCalls, + turn.toolOutputs, + ); + } + + /** + * Case B: flattens this attempt's input and its produced content into carry-over. Structured + * `tool_call_output` in the input (paired with the previous completed turn) is kept as-is; + * everything else (text input, model thinking/text, this attempt's tool calls/results) is + * transcribed into a single `` plain-text user message (includes all + * completed and incomplete messages, including partial thinking/text). If the input text is + * itself already a `` block (from a previous attempt or a previous run's + * carry-over), its content is unwrapped and merged in, keeping a single-level structure. + * + * TODO(multimodal): only text input is currently kept — `image_url` / `inline_data` input is + * lost during flatten (the `` structure has no corresponding transcription yet); + * multimodal carry-over support to be added later. + */ + private flattenCarryOver( + attemptInput: OmniMessage[], + assistantSegments: OmniMessage[], + toolCalls: OmniMessage[], + toolOutputs: OmniMessage[], + ): OmniMessage[] { + const structured = attemptInput.filter( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + ); + const textInputs = attemptInput.filter((m) => (m.payload as { type?: string }).type === "text"); + const flattened = userText( + this.buildTurnAbortedText(textInputs, assistantSegments, toolCalls, toolOutputs), + ); + // flatten is sent to the model only and not written to Trace (synthetic carry-over isn't + // persisted): resumption replay resends the discarded turn's **original input** as-is + // (best-effort), with no dependency on this synthetic message. + return [...structured, flattened]; + } + + /** Transcribes the interrupted turn's input, model thinking/text, and tool calls/results into a single `` plain-text block. */ + private buildTurnAbortedText( + textInputs: OmniMessage[], + assistantSegments: OmniMessage[], + toolCalls: OmniMessage[], + toolOutputs: OmniMessage[], + ): string { + const lines: string[] = [""]; + for (const m of textInputs) { + const t = (m.payload as TextPayload).text; + // If this text is itself already a synthetic block — a previous run's ``, + // or this turn's reconnect-appended `` — extract its inner lines and merge + // them in directly, avoiding layered nesting / unbounded growth (keeping a single-level + // structure). + const inner = unwrapSyntheticBlock(t); + if (inner !== null) { + if (inner) lines.push(inner); + } else { + lines.push(` ${t}`); + } + } + lines.push(...transcribeTurnLines(assistantSegments, toolCalls, toolOutputs)); + lines.push(""); + return lines.join("\n"); + } + + /** + * Assembles the reconnect retry input: the original input is kept as-is (structure and + * multimodal content preserved), with a `` text appended at the end carrying + * each failed attempt's thinking/text and tool calls/results produced so far; if nothing has + * been produced yet, it's just the original input. The synthetic message is sent to the model + * only and not written to Trace (same rule as flatten carry-over). + * Docs: /docs/agent-loop § "Automatic reconnect". + */ + private withRetriedTurns(input: OmniMessage[], failedTurns: TurnResult[]): OmniMessage[] { + const lines = failedTurns.flatMap((t) => + transcribeTurnLines(t.assistantSegments, t.toolCalls, t.toolOutputs), + ); + if (lines.length === 0) return input; + return [...input, userText(["", ...lines, ""].join("\n"))]; + } + + /** + * Trace writes are **best-effort**: observability should never interrupt the ReAct + * loop, so write failures only warn rather than throw. The first write after compaction first + * performs the deferred Trace rotation: splitting the file and opening it with session_meta. + */ + private async write(msg: OmniMessage): Promise { + if (!this.deps.trace) return; + if (this.pendingTraceRotation) { + this.pendingTraceRotation = false; + try { + if (this.deps.trace.rotate) await this.deps.trace.rotate(); + if (this.deps.sessionMeta) await this.deps.trace.write(this.deps.sessionMeta); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + process.stderr.write(`[trace] rotate failed: ${message}\n`); + } + } + try { + await this.deps.trace.write(msg); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + process.stderr.write(`[trace] write failed: ${message}\n`); + } + } +} + +/** + * If the text is an engine-synthesized whole block (`` or ``), + * strips the outer tags and returns the inner lines (may be an empty string); otherwise returns + * null. Both kinds of synthetic blocks use identical inner markup (thinking/text/tool_call/ + * tool_call_output), so it can be merged directly into a new block while keeping a single-level + * structure. + */ +function unwrapSyntheticBlock(text: string): string | null { + const m = /^<(turn_aborted|turn_retried)>\n?([\s\S]*?)\n?<\/\1>\s*$/.exec(text); + return m ? m[2]! : null; +} + +/** Transcribes the model's produced thinking/text and tool calls/results into tagged lines (shared by ``/``). */ +function transcribeTurnLines( + assistantSegments: OmniMessage[], + toolCalls: OmniMessage[], + toolOutputs: OmniMessage[], +): string[] { + const lines: string[] = []; + // The model's produced thinking/text (including partial segments finalized on interruption), + // written in production order. + for (const seg of assistantSegments) { + const p = seg.payload as { type?: string }; + if (p.type === "thinking") { + lines.push(` ${(seg.payload as ThinkingPayload).thinking}`); + } else if (p.type === "text") { + lines.push(` ${(seg.payload as TextPayload).text}`); + } + } + for (const tc of toolCalls) { + const p = tc.payload; + lines.push(` ${p.arguments}`); + } + for (const out of toolOutputs) { + const p = out.payload as ToolCallOutputPayload; + lines.push( + ` ${p.output}`, + ); + } + return lines; +} + +/** + * Extracts the summary within `` from compaction output; when the tag is + * missing, leniently uses the entire output as-is (not treated as a failure). Also used by + * Session resumption's "compaction closure" replay to + * reconstruct the `` pending input from the old Trace's compaction output. + */ +export function extractSummary(raw: string): string { + const match = /([\s\S]*?)<\/summary>/.exec(raw); + return (match ? match[1]! : raw).trim(); +} diff --git a/packages/core/src/environment/environment.ts b/packages/core/src/environment/environment.ts new file mode 100644 index 0000000..9310ba0 --- /dev/null +++ b/packages/core/src/environment/environment.ts @@ -0,0 +1,380 @@ +/** + * Environment —— executes approved tool calls inside the Workspace. + * + * Environment has no knowledge of any specific tool: it only assembles the tool names supported + * by ToolConfig into BuiltinTool instances (see `environment/tools/`), and dispatches execution + * by looking up the tool name. Adding a new built-in tool only requires implementing BuiltinTool + * and registering it — no changes to this file needed. Tool call **rendering** is not core's + * concern; it's handled by the CLI / Web frontend. + * + * The **framing and finalization** of the tool stream is handled uniformly by Environment: + * - Entering execution immediately emits `start`; the tool only needs to yield output deltas + * (its own start/stop are ignored); + * - Output is truncated online **front-to-back** by maxOutputLength (head is kept, forwarding + * stops once exceeded); the truncation marker, the tool's self-reported end marker + * (`ToolResult.note`, e.g. exit code — appended outside the truncation, never lost even when + * long output is truncated), and timeout/interruption/error markers are all emitted as part + * of the stream — **the content produced by concatenating streamed chunks matches the full + * message exactly**; + * - Nested session messages carrying an origin marker (e.g. forwarded from run_subagent) pass + * through unchanged, taking no part in this tool's output or finalization; + * - Argument parsing failures, unknown tool names, tool throws, and other exceptions all + * collapse into an explanatory, complete `tool_call_output` — never throws — and **output is + * never empty under any circumstance**. + * Docs: /docs/tools § "Execution contract". + */ +import { partialToolCallOutput, toolCallOutput } from "../omnimessage/index.js"; +import type { OmniMessage, StopReason } from "../omnimessage/index.js"; +import type { + EnvironmentConfig, + EnvironmentInterface, + ToolConfig, + ToolDefinition, + ToolExecutionRequest, + ToolPermission, +} from "../interfaces.js"; +import type { BuiltinTool, ToolResult } from "./tools/types.js"; +import { BUILTIN_TOOL_FACTORIES } from "./tools/registry.js"; +import { CommandSessionManager } from "./tools/command/index.js"; +import { SubagentSessionManager } from "./tools/subagent/index.js"; + +/** Default cap on tool output truncation (characters). */ +const DEFAULT_MAX_OUTPUT_LENGTH = 16000; + +/** Default timeout cap for a single tool call (milliseconds); <=0 disables it (every tool must be bound by timeoutMs). */ +const DEFAULT_TOOL_TIMEOUT_MS = 120000; + +/** Marker appended to the result when a tool is interrupted by the user. */ +const TOOL_ABORTED_NOTE = "[interrupted: tool aborted by user]"; + +/** Placeholder marker used when a tool produces no output at all (tool_call_output content is never empty). */ +const TOOL_EMPTY_NOTE = "[no output]"; + +/** + * Explanation for a failed argument JSON parse. The normal pipeline never reaches this: bad + * JSON already throws during AgentHub's parsing stage, and the LLM layer finalizes it as + * malformed for the engine to reconnect (see generative-model.ts) — it's never dispatched into + * Environment as a completed tool_call. This function is only a defensive fallback for the + * public interface. + */ +function describeArgumentsError(name: string, raw: string, err: unknown): string { + const detail = err instanceof Error ? err.message : String(err); + if (raw.trim() === "") { + return `Tool call "${name}" failed: the arguments field is empty. Re-issue the call with a complete JSON object.`; + } + return `Tool call "${name}" failed: the arguments are not valid JSON (${detail}). Re-issue the call with one complete, valid JSON object.`; +} + +/** Appends a marker after existing content: newline-joins if content is non-empty, otherwise just returns the marker. */ +function appendNote(base: string, note: string): string { + return base ? `${base}\n${note}` : note; +} + +/** The delta needed to stream out `note` on top of existing content `base` (includes separator, same basis as appendNote). */ +function noteSuffix(base: string, note: string): string { + return base ? `\n${note}` : note; +} + +export class Environment implements EnvironmentInterface { + private readonly workspaceDir: string; + private readonly toolConfig: ToolConfig; + /** Assembled built-in tools: tool name -> BuiltinTool. Only tools supported by the registry and present in config. */ + private readonly tools: Map; + /** Long-running command session registry: constructed within this Environment and shared between exec_command / input_command. */ + private readonly commandSessions: CommandSessionManager; + /** Background subagent session registry: constructed within this Environment and shared between run_subagent / input_subagent. */ + private readonly subagentSessions: SubagentSessionManager; + + constructor(config: EnvironmentConfig) { + this.workspaceDir = config.workspaceDir; + this.toolConfig = config.toolConfig; + this.tools = new Map(); + // The background session registry is created alongside Environment (one per Session) and + // injected into whichever tools need it; all sessions are finalized together on dispose. + // The vault environment variables are injected into child processes by the command session + // registry at spawn time. + this.commandSessions = new CommandSessionManager( + config.vault !== undefined ? { vault: config.vault } : {}, + ); + this.subagentSessions = new SubagentSessionManager(); + const services = { + ...config.services, + commandSessions: this.commandSessions, + subagentSessions: this.subagentSessions, + }; + // Assemble the tools supported by config into BuiltinTool instances; unrecognized tool + // names are skipped (neither exposed to the LLM nor executable). + for (const def of config.toolConfig.customTools) { + const factory = BUILTIN_TOOL_FACTORIES[def.name]; + if (factory) this.tools.set(def.name, factory(def, services)); + } + } + + /** Releases runtime resources held by Environment: finalizes all managed background sessions (command and subagent). Idempotent. */ + dispose(): void { + this.commandSessions.dispose(); + this.subagentSessions.dispose(); + } + + /** + * Lists tools available to the current Session, for context_engine to initialize GenerativeModel. + * Only lists tools that have been assembled (i.e. supported by the registry) — tool names + * unrecognized in config are not exposed to the LLM (consistent with the constructor); + * the definition (description/parameters) treats **the config entry as the single source of + * truth** — factories must not rewrite the definition at runtime; where a differentiated + * implementation is needed, use a separate explicit tool-name entry with a `forModel` + * annotation (e.g. read_image / describe_image). + * Only exposes `{name, description, parameters}`, dropping permission/maxOutputLength. + * MCP Server config flows into Environment via toolConfig; enumerating concrete MCP tools + * is left to a later adapter layer. + */ + async listTools(): Promise { + return this.toolConfig.customTools + .filter((tool) => this.tools.has(tool.name)) + .map((tool) => ({ + name: tool.name, + description: tool.description, + ...(tool.parameters !== undefined ? { parameters: tool.parameters } : {}), + })); + } + + /** Looks up a tool's permission level (for the frontend's permission-mode decisions); returns undefined for an unknown tool. */ + toolPermission(name: string): ToolPermission | undefined { + return this.toolConfig.customTools.find((t) => t.name === name)?.permission; + } + + /** + * Executes an approved tool call, streaming `partial_tool_call_output` and a final + * `tool_call_output`; nested messages carrying origin pass through unchanged. Dispatches by + * looking up the tool name; any exception collapses into an explanatory output — never throws. + * + * The priority for deciding stop_reason is: user interruption > timeout > tool throw > tool + * self-report. Interruption is determined by the `signal` held by Environment, and is + * compatible with both a tool self-reporting aborted and an AbortError raised by the + * interruption. An internal abort raised by a timeout does not count as a user interruption — + * it's finalized as failed, with the timeout reason written into the output. + * Docs: /docs/tools § "Execution contract". + */ + async *executeTool(request: ToolExecutionRequest): AsyncGenerator { + const payload = request.toolCall.payload; + // tool_call_id is passed through unchanged, so context_engine and the LLM can associate the + // request with its result. + const toolCallId = payload.tool_call_id; + const name = payload.name; + + // Every path is framed uniformly by Environment: entering execution emits start; the end + // uniformly emits stop + the full message. + yield partialToolCallOutput({ eventType: "start", toolCallId }); + + const tool = this.tools.get(name); + if (!tool) { + yield* emitFailure(toolCallId, `Unknown tool: ${name}`); + return; + } + + // Parse the tool's argument JSON; a parse failure collapses into an explanatory output + // (also streamed, so the frontend can render it). + let parsed: unknown; + try { + parsed = JSON.parse(payload.arguments); + } catch (err) { + yield* emitFailure(toolCallId, describeArgumentsError(name, payload.arguments, err)); + return; + } + + const args = + parsed !== null && typeof parsed === "object" ? (parsed as Record) : {}; + + const maxOutputLength = tool.definition.maxOutputLength ?? DEFAULT_MAX_OUTPUT_LENGTH; + const timeoutMs = tool.definition.timeoutMs ?? DEFAULT_TOOL_TIMEOUT_MS; + const signal = request.signal; + + // User interruption and tool timeout are merged into a single internal signal handed to the + // tool: either one triggers abortion of execution. + // The timeout constraint is enforced uniformly by Environment for all tools; the + // tool only needs to respond to signal. + const ac = new AbortController(); + if (signal?.aborted) ac.abort(); + const onAbort = (): void => ac.abort(); + signal?.addEventListener("abort", onAbort, { once: true }); + let timedOut = false; + const timer = + timeoutMs > 0 + ? setTimeout(() => { + timedOut = true; + ac.abort(); + }, timeoutMs) + : null; + timer?.unref?.(); + + // Consume the tool stream: content deltas are forwarded after online front-truncation; + // nested messages pass through; manual iteration to capture the generator's return value. + let streamed = ""; // Content forwarded so far (<= maxOutputLength) + let contentLen = 0; // Total length of content produced by the tool (including truncated/discarded parts) + let toolOutput: string | null = null; // Fallback: content basis when the tool produces a full message itself + let selfReported: StopReason | undefined; // Tool's self-reported stop reason (return value takes priority over the full message) + let selfNote: string | null = null; // Tool's self-reported end marker (e.g. exit code), appended outside truncation + let selfImages: string[] | undefined; // Tool's self-reported images (data URL), carried via a single streamed delta and the full message + let thrown: unknown = null; + const gen = tool.execute(args, { + workspaceDir: this.workspaceDir, + toolCallId, + signal: ac.signal, + // Pass through the parent's approve callback (run_subagent uses it so the child Session + // inherits the parent's approval mode; other tools ignore it). + ...(request.approve ? { approve: request.approve } : {}), + }); + try { + for (;;) { + const res = await gen.next(); + if (res.done) { + const result: ToolResult | void = res.value; + if (result?.stopReason) selfReported = result.stopReason; + if (result?.note) selfNote = result.note; + if (result?.images && result.images.length > 0) selfImages = result.images; + break; + } + const out = res.value; + if (out.origin && out.origin.length > 0) { + yield out; // Nested session message: pass through unchanged, not part of this tool's output/finalization + continue; + } + const p = out.payload as { + type?: string; + event_type?: string; + stop_reason?: string; + output?: string; + }; + if (p.type === "partial_tool_call_output") { + // Only takes delta content; start/stop are ignored (framing is uniformly handled by Environment). + if (p.event_type !== "delta" || !p.output) continue; + contentLen += p.output.length; + // maxOutputLength <= 0 means truncation is disabled (same semantics as timeoutMs). + const room = + maxOutputLength > 0 ? maxOutputLength - streamed.length : Number.POSITIVE_INFINITY; + if (room > 0) { + const chunk = p.output.length > room ? p.output.slice(0, room) : p.output; + streamed += chunk; + // Rebuild the delta: tool_call_id is uniformly enforced by Environment, never trusting the tool's own value. + yield partialToolCallOutput({ + eventType: "delta", + output: chunk, + toolCallId, + }); + } + } else if (p.type === "tool_call_output") { + // Fallback: if the tool still produces a full message, use it as the basis for content and stop reason (not needed under the new contract). + toolOutput = p.output ?? ""; + if (selfReported === undefined && p.stop_reason) { + selfReported = p.stop_reason as StopReason; + } + } else { + // Other message types without origin: protocol misuse, ignore and warn (keep the parent stream clean). + process.stderr.write( + `[penguin] tool "${name}" yielded unexpected message type "${p.type}"; ignored.\n`, + ); + } + } + } catch (err) { + // A tool throw also collapses into the uniform finalization: keep already-streamed content, don't discard produced output. + thrown = err; + } finally { + if (timer) clearTimeout(timer); + signal?.removeEventListener("abort", onAbort); + } + + // Uniform finalization. Content basis = the tool's self-produced full message (fallback + // path) or the already-forwarded delta; after front-truncating to the cap, the truncation + // marker and interruption/timeout/error markers are appended in turn, all made up via + // streamed deltas — streamed concatenation == the full message. + const contentBase = toolOutput ?? streamed; + const capped = + maxOutputLength > 0 && contentBase.length > maxOutputLength + ? contentBase.slice(0, maxOutputLength) + : contentBase; + const truncated = capped.length < contentBase.length || contentLen > streamed.length; + + const aborted = + signal?.aborted === true || + (!timedOut && + (selfReported === "aborted" || + (thrown as { name?: string } | null)?.name === "AbortError")); + let stopReason: StopReason; + const notes: string[] = []; + if (truncated) { + notes.push(`[output truncated: exceeded ${maxOutputLength} chars]`); + } + // The tool's self-reported end marker (e.g. exit code): appended outside the truncation — + // if treated as a content delta it would get cut off once long output hits the cap, and the + // model would misread a command that failed after printing lots of output as successful. + if (selfNote) { + notes.push(selfNote); + } + if (aborted) { + stopReason = "aborted"; + notes.push(TOOL_ABORTED_NOTE); + } else if (timedOut) { + stopReason = "failed"; + notes.push(`[tool timeout: exceeded ${timeoutMs}ms]`); + } else if (thrown != null) { + stopReason = "failed"; + notes.push(`[tool error] ${thrown instanceof Error ? thrown.message : String(thrown)}`); + } else { + stopReason = selfReported ?? "completed"; + } + // The tool's reply must never be empty: an empty tool_result leaves the model unable to + // tell "silent success" apart from "call failed", and some Providers outright reject empty + // content blocks. + if (capped === "" && notes.length === 0) { + notes.push(TOOL_EMPTY_NOTE); + } + const noteText = notes.join("\n"); + const fullOutput = noteText ? appendNote(capped, noteText) : capped; + + // Compensating content delta: if nothing was streamed, emit the whole thing at once; on the + // fallback path, emit the portion of the full message beyond the already-streamed prefix + // (if the tool is internally inconsistent, the full message wins — no further reconciliation). + let compensation = ""; + if (streamed === "") compensation = capped; + else if (toolOutput !== null && capped.startsWith(streamed)) { + compensation = capped.slice(streamed.length); + } + if (compensation) { + yield partialToolCallOutput({ + eventType: "delta", + output: compensation, + toolCallId, + }); + } + if (noteText) { + yield partialToolCallOutput({ + eventType: "delta", + output: noteSuffix(capped, noteText), + toolCallId, + }); + } + // Images are made up via streaming: images are not delta'd — a single delta carries them + // all at once right before stop, and the full message carries them again — satisfying + // "streamed concatenation == full message" the same way text does (truncation only applies + // to text, never touches images). + // Only carried on normal completion; interruption/timeout/error paths carry no images, to keep finalization simple. + const images = stopReason === "completed" ? selfImages : undefined; + if (images) { + yield partialToolCallOutput({ eventType: "delta", toolCallId, images }); + } + yield partialToolCallOutput({ eventType: "stop", toolCallId, stopReason }); + yield toolCallOutput({ + output: fullOutput, + toolCallId, + stopReason, + ...(images ? { images } : {}), + }); + } +} + +/** Upfront failure (unknown tool/argument parse failure): delta(explanation) -> stop -> full failed output (start already emitted by the caller). */ +function* emitFailure(toolCallId: string, message: string): Generator { + yield partialToolCallOutput({ eventType: "delta", output: message, toolCallId }); + yield partialToolCallOutput({ eventType: "stop", toolCallId, stopReason: "failed" }); + yield toolCallOutput({ output: message, toolCallId, stopReason: "failed" }); +} diff --git a/packages/core/src/environment/index.ts b/packages/core/src/environment/index.ts new file mode 100644 index 0000000..00e58d7 --- /dev/null +++ b/packages/core/src/environment/index.ts @@ -0,0 +1,14 @@ +/** + * Environment module barrel — exports the environment interface implementation and builtin tool abstractions. + */ +export { Environment } from "./environment.js"; +export type { BuiltinTool, ToolExecutionContext } from "./tools/types.js"; +export { BUILTIN_TOOL_FACTORIES } from "./tools/registry.js"; +export type { BuiltinToolFactory } from "./tools/registry.js"; +export { createExecCommandTool, EXEC_COMMAND_NAME } from "./tools/exec-command.js"; +export { createInputCommandTool, INPUT_COMMAND_NAME } from "./tools/input-command.js"; +export { createSubagentTool, SUBAGENT_NAME } from "./tools/run-subagent.js"; +export { createInputSubagentTool, INPUT_SUBAGENT_NAME } from "./tools/input-subagent.js"; +export { CommandSessionManager, ManagedSession } from "./tools/command/index.js"; +export type { ProcessExit, SpawnOptions } from "./tools/command/index.js"; +export { SubagentSessionManager, ManagedSubagentSession } from "./tools/subagent/index.js"; diff --git a/packages/core/src/environment/tools/background/capped-buffer.ts b/packages/core/src/environment/tools/background/capped-buffer.ts new file mode 100644 index 0000000..ba25d92 --- /dev/null +++ b/packages/core/src/environment/tools/background/capped-buffer.ts @@ -0,0 +1,44 @@ +/** + * CappedTextBuffer — capacity-capped unread-text buffer shared by background sessions. + * + * When over capacity, drops the oldest content (keeping the tail) and tallies the dropped count; + * `drain()` prefixes a marker noting the drop count when taking all unread content (guards + * against a chatty background process / sub-agent blowing up memory, see each session class's + * capacity constant). + */ +export class CappedTextBuffer { + private text = ""; + private omitted = 0; // Characters dropped due to the capacity cap, not yet read + + /** `dropLabel` is used in the drop-marker text, e.g. "earlier output" / "earlier subagent output". */ + constructor( + private readonly cap: number, + private readonly dropLabel: string, + ) {} + + get isEmpty(): boolean { + return this.text.length === 0 && this.omitted === 0; + } + + append(chunk: string): void { + this.text += chunk; + if (this.text.length > this.cap) { + const drop = this.text.length - this.cap; + this.text = this.text.slice(drop); // Keep the newest (tail), drop the oldest + this.omitted += drop; + } + } + + /** Takes the current unread content (including the drop marker); clears the buffer. */ + drain(): string { + if (this.isEmpty) return ""; + const b = this.text; + this.text = ""; + if (this.omitted > 0) { + const n = this.omitted; + this.omitted = 0; + return `[... ${n} chars of ${this.dropLabel} dropped ...]\n${b}`; + } + return b; + } +} diff --git a/packages/core/src/environment/tools/background/index.ts b/packages/core/src/environment/tools/background/index.ts new file mode 100644 index 0000000..6f55a7d --- /dev/null +++ b/packages/core/src/environment/tools/background/index.ts @@ -0,0 +1,8 @@ +/** + * Barrel for shared background-session infrastructure. + */ +export { BackgroundRegistry } from "./registry.js"; +export type { BackgroundTask } from "./registry.js"; +export { clampYield } from "./limits.js"; +export { WakeSignal } from "./wake-signal.js"; +export { CappedTextBuffer } from "./capped-buffer.js"; diff --git a/packages/core/src/environment/tools/background/limits.ts b/packages/core/src/environment/tools/background/limits.ts new file mode 100644 index 0000000..2e7a745 --- /dev/null +++ b/packages/core/src/environment/tools/background/limits.ts @@ -0,0 +1,23 @@ +/** + * Yield-time clamping shared by background-session tools. + * + * `yield_time_ms` is a soft budget for a single tool call: "wait at most until the session ends + * or this duration expires" (expiry yields, it's not a failure). Only a lower bound is set: a + * wait that's too short isn't meaningful and just adds round trips. The upper bound is no longer + * an independent constant — it's derived from the tool's own `timeoutMs` (with a reserved + * margin, so the yield happens before the Environment's timeout fallback fires); no upper bound + * is set when `timeoutMs <= 0` (disabled). + */ + +/** Lower bound for yield time (ms). */ +export const MIN_YIELD_MS = 250; +/** Margin (ms) reserved between the yield upper bound and the tool's `timeoutMs`: the yield must happen before the timeout fallback. */ +const TIMEOUT_MARGIN_MS = 1_000; + +/** Clamps the raw argument to `[MIN_YIELD_MS, timeoutMs - margin]`; falls back to `fallback` if not a number, no upper bound when `timeoutMs <= 0`. */ +export function clampYield(raw: unknown, fallback: number, timeoutMs?: number): number { + const n = typeof raw === "number" && Number.isFinite(raw) ? raw : fallback; + const lower = Math.max(n, MIN_YIELD_MS); + if (timeoutMs === undefined || timeoutMs <= 0) return lower; + return Math.min(lower, Math.max(timeoutMs - TIMEOUT_MARGIN_MS, MIN_YIELD_MS)); +} diff --git a/packages/core/src/environment/tools/background/registry.ts b/packages/core/src/environment/tools/background/registry.ts new file mode 100644 index 0000000..f7820f0 --- /dev/null +++ b/packages/core/src/environment/tools/background/registry.ts @@ -0,0 +1,174 @@ +/** + * BackgroundRegistry —— generic registry for background sessions, shared by command sessions + * and subagent sessions. + * + * Responsibilities: allocating and managing background session ids, enforcing the concurrency + * cap, reclaiming idle sessions, uniform finalization when the Session/Environment ends, and a + * hard kill on process 'exit' as a fallback (JS has no destructors, so cleanup must be explicit). + * Background sessions living for days at a time are a legitimate form; idle reclamation is only + * a leak fallback — sessions unaccessed for longer than `IDLE_TTL_MS` are finalized by a + * periodic sweep. + * + * The two concurrency-cap strategies are expressed by how `makeRoom` is called: + * - Command sessions: if full at registration time, prefer evicting exited sessions, otherwise + * evict LRU (killing a background process is an acceptable cost); + * - Subagent sessions: `makeRoom` is called before launch (only evicting completed, idle ones); + * if there's no room, spawning is rejected — evicting a running subagent is equivalent to + * discarding in-progress work, which is semantically unacceptable. + * + * No lock is needed under the single-threaded event loop; but note the registry may change + * across an `await`, so check before using an entry. + * Docs: /docs/tools § "Background session caps". + */ +import { randomUUID } from "node:crypto"; + +/** Idle reclamation TTL (milliseconds): a session unaccessed for longer than this is treated as a leak and reclaimed. */ +const IDLE_TTL_MS = 10 * 24 * 60 * 60_000; // 10 days +/** Idle reclamation check interval (milliseconds): TTL is measured in days, so an hourly sweep is sufficient. */ +const IDLE_SWEEP_MS = 60 * 60_000; + +/** Minimal contract a background session must satisfy to be managed by the registry. */ +export interface BackgroundTask { + /** Timestamp of the most recent access (used for LRU eviction); refreshed by the registry on register/get. */ + lastUsed: number; + /** Whether the session is still running (determines eviction priority). */ + running: boolean; + /** Asynchronous finalization (SIGTERM -> SIGKILL / abort); idempotent. */ + kill(): void; + /** Synchronous hard kill (process 'exit' fallback path: the event loop has stopped, timers are unavailable). */ + killHard(): void; +} + +// process 'exit' fallback: use a single module-level listener to manage all registries, avoiding +// each Session adding its own listener and triggering EventEmitter's MaxListeners warning. +const LIVE_REGISTRIES = new Set>(); +let exitHookInstalled = false; +function ensureExitHook(): void { + if (exitHookInstalled) return; + exitHookInstalled = true; + process.on("exit", () => { + for (const r of LIVE_REGISTRIES) r.killAllHard(); + }); +} + +export class BackgroundRegistry { + private readonly tasks = new Map(); + private readonly idPrefix: string; + private readonly maxTasks: number; + private readonly reapTimer: ReturnType; + private disposed = false; + + constructor(opts: { idPrefix: string; maxTasks: number }) { + this.idPrefix = opts.idPrefix; + this.maxTasks = opts.maxTasks; + LIVE_REGISTRIES.add(this as unknown as BackgroundRegistry); + ensureExitHook(); + this.reapTimer = setInterval(() => this.reapIdle(), IDLE_SWEEP_MS); + this.reapTimer.unref?.(); + } + + get size(): number { + return this.tasks.size; + } + + /** + * Makes room for a new session. Returns true immediately if not full; when full, evicts per + * `evictRunning`: + * - false (subagent): only evicts the least-recently-used **completed** session; if all are + * running, returns false (the caller rejects spawning); + * - true (command): evicts exited sessions first, otherwise LRU-evicts a running one. + */ + makeRoom(evictRunning: boolean): boolean { + if (this.tasks.size < this.maxTasks) return true; + let lruId: string | null = null; + let lruUsed = Number.POSITIVE_INFINITY; + for (const [id, t] of this.tasks) { + if (!t.running) { + this.remove(id); // Prefer evicting sessions that have already ended + return true; + } + if (t.lastUsed < lruUsed) { + lruUsed = t.lastUsed; + lruId = id; + } + } + if (!evictRunning) return false; + if (lruId) this.remove(lruId); + return this.tasks.size < this.maxTasks; + } + + /** + * Registers a session, allocating and returning a unique id (`-xxxxxxxx`). The caller + * must first free up room via `makeRoom`. `preferredSuffix` is the preferred id suffix (e.g. + * the tail of a child Session id, so the tool handle correlates with the message origin/ + * frontend nesting label); falls back to random if omitted or on collision. + */ + register(task: T, preferredSuffix?: string): string { + this.ensureActive(); + let id = preferredSuffix ? `${this.idPrefix}-${preferredSuffix}` : this.randomId(); + while (this.tasks.has(id)) id = this.randomId(); + task.lastUsed = Date.now(); + this.tasks.set(id, task); + return id; + } + + private randomId(): string { + return `${this.idPrefix}-${randomUUID().replace(/-/g, "").slice(0, 8)}`; + } + + /** Looks up a session by id and refreshes its access time; returns undefined if not found. */ + get(id: string): T | undefined { + if (this.disposed) return undefined; + const t = this.tasks.get(id); + if (t) t.lastUsed = Date.now(); + return t; + } + + /** Removes a session from the registry and finalizes it. */ + remove(id: string): void { + const t = this.tasks.get(id); + if (!t) return; + this.tasks.delete(id); + t.kill(); + } + + /** Kills and clears all sessions (called when the Session/Environment ends). */ + killAll(): void { + for (const t of this.tasks.values()) t.kill(); + this.tasks.clear(); + } + + /** Synchronously hard-kills all sessions (process 'exit' fallback path). */ + killAllHard(): void { + for (const t of this.tasks.values()) t.killHard(); + this.tasks.clear(); + } + + /** Disposes: removes the fallback registration and kills all sessions. Idempotent. */ + dispose(): void { + if (this.disposed) return; + this.disposed = true; + clearInterval(this.reapTimer); + LIVE_REGISTRIES.delete(this as unknown as BackgroundRegistry); + this.killAll(); + } + + /** Whether the registry has been disposed (the host Session has ended). */ + get isDisposed(): boolean { + return this.disposed; + } + + /** Reclaims sessions idle for longer than `IDLE_TTL_MS` (leak fallback, triggered by the periodic sweep). */ + private reapIdle(): void { + const cutoff = Date.now() - IDLE_TTL_MS; + for (const [id, t] of this.tasks) { + if (t.lastUsed <= cutoff) this.remove(id); + } + } + + private ensureActive(): void { + if (this.disposed) { + throw new Error("background session registry disposed"); + } + } +} diff --git a/packages/core/src/environment/tools/background/wake-signal.ts b/packages/core/src/environment/tools/background/wake-signal.ts new file mode 100644 index 0000000..fabbb4d --- /dev/null +++ b/packages/core/src/environment/tools/background/wake-signal.ts @@ -0,0 +1,43 @@ +/** + * WakeSignal —— a single wakeup point shared by background sessions. + * + * Producer events (data arrival / run finished / new approval request) call `notify()`; + * waiters use `wait(ms)` to wait for "woken up" or expiry, whichever comes first. `notify` + * swaps in a new promise before resolving the old one, so a waiter that wakes up just + * re-checks state — it never misses an event that immediately follows. + */ +export class WakeSignal { + private promise!: Promise; + private resolve!: () => void; + + constructor() { + this.arm(); + } + + private arm(): void { + this.promise = new Promise((resolve) => { + this.resolve = resolve; + }); + } + + /** Wakes up all waiters: swaps in a new promise before resolving the old one (avoids missing an event that immediately follows). */ + notify(): void { + const r = this.resolve; + this.arm(); + r(); + } + + /** Waits for "woken up" or `ms` to elapse, whichever comes first. */ + async wait(ms: number): Promise { + let timer: ReturnType | null = null; + const timeout = new Promise((resolve) => { + // wait is part of an active operation: the timer needs to keep the process alive, and is cleaned up immediately below after notify. + timer = setTimeout(resolve, ms); + }); + try { + await Promise.race([this.promise, timeout]); + } finally { + if (timer) clearTimeout(timer); + } + } +} diff --git a/packages/core/src/environment/tools/command/index.ts b/packages/core/src/environment/tools/command/index.ts new file mode 100644 index 0000000..d94f444 --- /dev/null +++ b/packages/core/src/environment/tools/command/index.ts @@ -0,0 +1,11 @@ +/** + * Barrel for the long-running command session module. + */ +export { CommandSessionManager } from "./session-manager.js"; +export { ManagedSession, resultForExit } from "./session.js"; +export type { ProcessExit, SpawnOptions } from "./session.js"; +export { + DEFAULT_EXEC_YIELD_MS, + DEFAULT_WRITE_YIELD_MS, + DEFAULT_EMPTY_POLL_YIELD_MS, +} from "./limits.js"; diff --git a/packages/core/src/environment/tools/command/limits.ts b/packages/core/src/environment/tools/command/limits.ts new file mode 100644 index 0000000..cc338ec --- /dev/null +++ b/packages/core/src/environment/tools/command/limits.ts @@ -0,0 +1,15 @@ +/** + * Default yield durations for long-running command sessions. + * + * `yield_time_ms` is the soft budget for a tool call to "wait at most until the command ends or + * this duration elapses" (yielding on expiry is not a failure); see `../background/limits.ts` + * for the clamping logic: it only sets a floor, the ceiling is derived from the tool's own + * `timeoutMs`. + */ + +/** Default wait duration (milliseconds) for `exec_command` starting a command. */ +export const DEFAULT_EXEC_YIELD_MS = 60_000; +/** Default wait duration (milliseconds) for `input_command` when there's a write. */ +export const DEFAULT_WRITE_YIELD_MS = 250; +/** Default wait duration (milliseconds) for `input_command` on an empty poll. */ +export const DEFAULT_EMPTY_POLL_YIELD_MS = 5_000; diff --git a/packages/core/src/environment/tools/command/session-manager.ts b/packages/core/src/environment/tools/command/session-manager.ts new file mode 100644 index 0000000..1d6238a --- /dev/null +++ b/packages/core/src/environment/tools/command/session-manager.ts @@ -0,0 +1,83 @@ +/** + * CommandSessionManager — registry and lifecycle management for long-running command sessions. + * + * Constructed by Environment (one per Session), injected via services and shared by the + * `exec_command` and `input_command` tools. Registry responsibilities (id allocation, concurrency + * cap, dispose, process 'exit' fallback) are handled by the generic `BackgroundRegistry` (shared + * with subagent sessions, see `../background/registry.ts`); this class only retains + * command-domain logic: spawning processes and assembling the child process environment (vault + * injection + hardening). + * Docs: /docs/tools § "Background session caps". + */ +import { ManagedSession } from "./session.js"; +import { BackgroundRegistry } from "../background/index.js"; + +/** Concurrent managed-session cap: evicts once exceeded (exited sessions first, otherwise LRU — killing a background process has bounded cost). */ +const MAX_SESSIONS = 64; + +/** + * Hardening overrides applied to the child process environment: suppresses editor/credential + * prompts/pagers/color etc. that could interact, avoiding a command hanging while waiting for + * input. `GIT_EDITOR=true` prevents `git commit`/`rebase -i` from popping an editor; + * `GIT_TERMINAL_PROMPT=0` prevents git from interactively asking for credentials; in pipe mode, + * git and similar tools already auto-disable the pager, so the `PAGER` entries are just an extra + * safeguard. + */ +const HARDENED_ENV: NodeJS.ProcessEnv = { + GIT_EDITOR: "true", + GIT_TERMINAL_PROMPT: "0", + TERM: "dumb", + NO_COLOR: "1", + PAGER: "cat", + GIT_PAGER: "cat", +}; + +export class CommandSessionManager { + private readonly registry = new BackgroundRegistry({ + idPrefix: "proc", + maxTasks: MAX_SESSIONS, + }); + + /** Agent vault environment variables: injected into the child process on every spawn (values never enter the model context, only the environment). */ + private readonly vault: Record; + + constructor(opts?: { vault?: Record }) { + this.vault = opts?.vault ?? {}; + } + + /** Starts a command, returning an **unregistered** session (no process_id yet). */ + spawn(opts: { cmd: string; cwd: string }): ManagedSession { + if (this.registry.isDisposed) { + throw new Error("command session manager disposed"); + } + return new ManagedSession({ + cmd: opts.cmd, + cwd: opts.cwd, + // Spread order is priority: vault overrides host variables of the same name, but must + // come before HARDENED_ENV — the hardening entries (GIT_EDITOR/PAGER etc. that prevent + // interactive hangs) must never be overridable by vault. + env: { ...process.env, ...this.vault, ...HARDENED_ENV }, + }); + } + + /** Registers a still-running session as a background process, allocating and returning a unique `process_id`. */ + register(session: ManagedSession): string { + this.registry.makeRoom(true); + return this.registry.register(session); + } + + /** Looks up a session by process_id and refreshes its access time; returns undefined if it doesn't exist. */ + get(processId: string): ManagedSession | undefined { + return this.registry.get(processId); + } + + /** Removes from the registry and cleans up the process group (called after the session exits). */ + remove(processId: string): void { + this.registry.remove(processId); + } + + /** Disposes: removes the fallback registration and kills all sessions (the process 'exit' fallback is hooked up by the registry itself). Idempotent. */ + dispose(): void { + this.registry.dispose(); + } +} diff --git a/packages/core/src/environment/tools/command/session.ts b/packages/core/src/environment/tools/command/session.ts new file mode 100644 index 0000000..6cf1dd4 --- /dev/null +++ b/packages/core/src/environment/tools/command/session.ts @@ -0,0 +1,228 @@ +/** + * ManagedSession — runtime state and collection logic for a single command session. + * + * Spawns the process with `bash -lc `, with stdout/stderr going through plain pipes (no + * native dependency, clean output; an interactive program that detects no TTY falls back to + * non-interactive mode, which parses more cleanly for the Agent anyway). `detached` makes the + * child process the process-group leader, so both Ctrl-C and killing the whole group rely on + * **process-group signals** (sending a signal to `-pid` also reaches background child processes). + * + * Key semantics: + * - **Termination is determined by the foreground process exiting (the exit event, waitpid + * semantics), not by waiting for stream EOF**: background child processes that inherit the + * pipe don't hold things up; + * - `collect(yieldMs)` **streams** output deltas within the budget: data is yielded as soon as it + * arrives, without waiting for the window to end; if the command exits mid-window, the trailing + * output is yielded along with it (with a capped drain window); if it's still running once the + * window expires, whatever output exists is yielded and collection ends, with the process + * switching to background; if `signal` aborts, whatever output exists is yielded and collection + * ends immediately; + * - Unread output has a cap (memory safety); when exceeded, the oldest part is dropped and + * counted, with a marker shown on read; + * - `kill()` sends SIGTERM to the process group, then SIGKILL after a grace period, reaping any + * leftover background child processes; idempotent. + */ +import { spawn, type ChildProcess } from "node:child_process"; +import type { ToolResult } from "../types.js"; +import { CappedTextBuffer, WakeSignal } from "../background/index.js"; + +/** Process-group semantics are available on POSIX; Windows falls back to signaling the child process directly. */ +const SUPPORTS_PROCESS_GROUP = process.platform !== "win32"; + +/** Extra wait cap (ms) after the command exits to collect trailing output: enough to drain the last flush, without hanging. */ +const POST_EXIT_DRAIN_MS = 50; +/** Capacity cap (characters) for a single session's unread output: prevents a chatty background process from blowing up memory. */ +const OUTPUT_BUFFER_CAP = 1024 * 1024; // 1 MiB +/** + * Grace period (ms) before escalating from SIGTERM to SIGKILL: gives a process that needs to + * clean up (flush data, remove temp files) some time. The timer is unref'd, so it won't hold up + * the host process from exiting; the process-exit path sends SIGKILL directly as a fallback. + */ +const SIGKILL_GRACE_MS = 1_000; + +/** Foreground process exit info. At most one of `code`/`signal` is set (consistent with Node child's exit event). */ +export interface ProcessExit { + code: number | null; + signal: NodeJS.Signals | null; +} + +/** Arguments required to start a command. */ +export interface SpawnOptions { + /** Command string handed to `bash -lc`. */ + cmd: string; + /** Working directory (absolute path). */ + cwd: string; + /** Child process environment variables (the caller has already injected hardening entries like PAGER/TERM). */ + env: NodeJS.ProcessEnv; +} + +export class ManagedSession { + /** Timestamp of the last access (used for LRU / idle reaping). */ + lastUsed: number = Date.now(); + + private readonly child: ChildProcess; + private readonly buffer = new CappedTextBuffer(OUTPUT_BUFFER_CAP, "earlier output"); + private exited = false; + private exitInfo: ProcessExit | null = null; + private spawnError: Error | null = null; + private killed = false; + private killTimer: ReturnType | null = null; + // Single wake point: data arrival / process exit / spawn error all wake a waiting collect through it. + private readonly wakeSignal = new WakeSignal(); + + constructor(opts: SpawnOptions) { + this.child = spawn("bash", ["-lc", opts.cmd], { + cwd: opts.cwd, + env: opts.env, + detached: SUPPORTS_PROCESS_GROUP, // Become the process-group leader, so the whole group can be signaled + stdio: ["pipe", "pipe", "pipe"], + }); + this.child.stdout?.setEncoding("utf8"); + this.child.stderr?.setEncoding("utf8"); + // stdin may already be closed by the command before input_command writes to it; + // EPIPE/ERR_STREAM_DESTROYED are an expected race and must not bubble up to the host process + // as an unhandled error. + this.child.stdin?.on("error", () => {}); + this.child.stdout?.on("data", (c: string) => this.handleData(c)); + this.child.stderr?.on("data", (c: string) => this.handleData(c)); + // exit follows waitpid semantics: it fires as soon as bash exits, without waiting for + // stdout/stderr pipe EOF — background child processes that inherit and hold the pipe open + // won't hold up termination. + this.child.on("exit", (code, signal) => this.handleExit({ code, signal })); + this.child.on("error", (err) => this.handleError(err)); + } + + /** Signals the process group; ignores the case where the process/group has already exited (ESRCH). */ + private signalGroup(sig: NodeJS.Signals): void { + try { + if (SUPPORTS_PROCESS_GROUP && typeof this.child.pid === "number" && this.child.pid > 0) { + process.kill(-this.child.pid, sig); // Negative pid = the whole process group + } else { + this.child.kill(sig); + } + } catch { + // ESRCH etc., ignored. + } + } + + private handleData(chunk: string): void { + this.buffer.append(chunk); + this.wakeSignal.notify(); + } + private handleExit(exit: ProcessExit): void { + if (this.exited) return; + this.exited = true; + this.exitInfo = exit; + this.wakeSignal.notify(); + } + private handleError(err: Error): void { + if (this.exited) return; + this.spawnError = err; + this.exited = true; // A spawn failure is also treated as a terminal state + this.wakeSignal.notify(); + } + + /** Whether the command is still running (hasn't exited, spawn hasn't failed). */ + get running(): boolean { + return !this.exited; + } + get exit(): ProcessExit | null { + return this.exitInfo; + } + get error(): Error | null { + return this.spawnError; + } + + /** + * Streams output deltas within `yieldMs` (data is yielded as soon as it arrives). Once done, + * the terminal state is determined via `running`/`exit`/`error`: + * - Exits mid-window -> the trailing output is yielded along with it (extra ≤POST_EXIT_DRAIN_MS drain); + * - Still running once the window expires -> whatever output exists is yielded and collection ends, with the process switching to background; + * - `signal` aborts -> whatever output exists is yielded and collection ends immediately (the process isn't killed; the caller decides whether to keep it). + */ + async *collect(yieldMs: number, signal?: AbortSignal): AsyncGenerator { + const start = Date.now(); + const onAbort = (): void => this.wakeSignal.notify(); + signal?.addEventListener("abort", onAbort, { once: true }); + try { + // Phase one: running, data is yielded as soon as it arrives, until exit / abort / yield expires. + while (!this.exited) { + const chunk = this.buffer.drain(); + if (chunk) yield chunk; + if (signal?.aborted) return; + const remaining = yieldMs - (Date.now() - start); + if (remaining <= 0) { + const tail = this.buffer.drain(); + if (tail) yield tail; + return; // Still running -> yield + } + // Re-check the predicate before sleeping: data that arrives while `yield` is suspended + // wakes at a point before this wait begins, and would otherwise be missed. + if (!this.buffer.isEmpty) continue; + await this.wakeSignal.wait(remaining); + } + // Phase two: already exited (or spawn failed) -> drain the trailing output, with a cap. + const head = this.buffer.drain(); + if (head) yield head; + const drainStart = Date.now(); + for (;;) { + if (!this.buffer.isEmpty) { + yield this.buffer.drain(); + continue; + } + const left = POST_EXIT_DRAIN_MS - (Date.now() - drainStart); + if (left <= 0) break; + await this.wakeSignal.wait(left); + if (this.buffer.isEmpty) break; // Woke with no new data (or timed out) -> draining is done + } + const tail = this.buffer.drain(); + if (tail) yield tail; + } finally { + signal?.removeEventListener("abort", onAbort); + } + } + + write(chars: string): void { + this.lastUsed = Date.now(); + try { + if (!this.child.stdin || this.child.stdin.destroyed) return; + this.child.stdin.write(chars, () => {}); + } catch { + // stdin may already be closed, ignored. + } + } + interrupt(): void { + this.lastUsed = Date.now(); + this.signalGroup("SIGINT"); + } + + /** Closes out: sends SIGTERM to the process group, then SIGKILL after a grace period (reaping leftover background child processes); idempotent. */ + kill(): void { + if (this.killed) return; + this.killed = true; + this.signalGroup("SIGTERM"); + // Unconditionally escalates to SIGKILL: the foreground has exited but background child + // processes may still be around; killpg on an already-vanished group is ESRCH (harmless). + this.killTimer = setTimeout(() => this.signalGroup("SIGKILL"), SIGKILL_GRACE_MS); + this.killTimer.unref?.(); + } + + /** Synchronous hard kill (process 'exit' fallback: the event loop has already stopped at this point, so timers aren't available). */ + killHard(): void { + this.killed = true; + if (this.killTimer) { + clearTimeout(this.killTimer); + this.killTimer = null; + } + this.signalGroup("SIGKILL"); + } +} + +/** Converts exit info into a tool result (the terminal marker is appended via `note`, outside the truncation, so it isn't lost with long output). */ +export function resultForExit(exit: ProcessExit | null): ToolResult { + if (!exit) return { stopReason: "completed" }; + if (exit.signal) return { stopReason: "failed", note: `[terminated by signal ${exit.signal}]` }; + if (exit.code !== 0) + return { stopReason: "failed", note: `[exit code: ${exit.code ?? "unknown"}]` }; + return { stopReason: "completed" }; +} diff --git a/packages/core/src/environment/tools/describe-image.ts b/packages/core/src/environment/tools/describe-image.ts new file mode 100644 index 0000000..abb9baa --- /dev/null +++ b/packages/core/src/environment/tools/describe-image.ts @@ -0,0 +1,115 @@ +/** + * describe_image —— image-proxy-read tool, the text-only-model variant of read_image + * (`forModel: "text-only"`, see default-config.ts): the image itself is never fed back into + * the session model (some providers flatly 400 on a tool_result carrying an image); instead it + * is sent, together with the caller-supplied `prompt`, in a single one-off request to the + * Project-configured vision model (`vision_model`), and the vision model's text answer is + * returned as the tool's output. + * + * The tool definition (description/parameters, including `prompt`) comes entirely from the + * config entry; this implementation does no runtime rewriting — which tool is used for which + * model class is declared by the entry's `forModel` annotation, so the config file is what you get. + * + * Behavioral contract (shared with read_image): `source` supports http(s) URLs and local paths; + * validation/size limits are reused from `loadImage`; on failure, outputs explanatory text and + * finishes with `failed`, never throws; on interruption, only reports `aborted`. Messages from + * the internal one-off request never enter the parent session stream (no origin, not leaked out). + * Docs: /docs/tools § "Image tools". + */ +import { imageUrlMessage, partialToolCallOutput, userText } from "../../omnimessage/index.js"; +import type { OmniMessage } from "../../omnimessage/index.js"; +import type { LLMOutcome, ToolDefinitionConfig, VisionDescriberService } from "../../interfaces.js"; +import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js"; +import { formatSize, loadImage } from "./read-image.js"; + +/** Tool name constant (used only inside this tool module, not exposed to Environment). */ +export const DESCRIBE_IMAGE_NAME = "describe_image"; + +/** Default question used when the caller doesn't supply a prompt. */ +const DEFAULT_PROMPT = + "Describe this image in detail, including any visible text, numbers, UI elements and layout."; + +/** Constructs the describe_image tool: definition (description/parameters) is taken as-is from the config entry. */ +export function createDescribeImageTool( + definition: ToolDefinitionConfig, + describer: VisionDescriberService, +): BuiltinTool { + return { + name: DESCRIBE_IMAGE_NAME, + definition, + async *execute( + args: Record, + ctx: ToolExecutionContext, + ): AsyncGenerator { + const { toolCallId, signal } = ctx; + const delta = (output: string): OmniMessage => + partialToolCallOutput({ eventType: "delta", output, toolCallId }); + + const source = args["source"]; + if (typeof source !== "string" || source.length === 0) { + yield delta('Missing required argument "source" for describe_image.'); + return { stopReason: "failed" }; + } + if (describer.modelId === null || describer.createLLM === undefined) { + yield delta( + "No vision model is configured for this project. The current model does not accept " + + "images; ask the user to pick a vision model in the model settings (vision_model) " + + "to enable image reading.", + ); + return { stopReason: "failed" }; + } + + const res = await loadImage(source, ctx.workspaceDir, signal); + if (!res.ok) { + if (res.reason === "aborted") return { stopReason: "aborted" }; + yield delta(res.message); + return { stopReason: "failed" }; + } + + const prompt = + typeof args["prompt"] === "string" && args["prompt"].trim().length > 0 + ? args["prompt"] + : DEFAULT_PROMPT; + const dataUrl = `data:${res.mime};base64,${res.bytes.toString("base64")}`; + + // One-off vision model request: prompt + image are merged into a single user message; + // its text deltas (partial_text delta) are forwarded in real time as this tool's own + // output delta — the description streams out piece by piece, not buffered as a whole. + // Partial concatenation == the full message (see generative-model.ts), so the full text + // is not forwarded again; other messages like thinking/token_usage are ignored, never + // leaked into the parent session. + const llm = describer.createLLM(); + const gen = llm.streamGenerate({ + newMessages: [userText(prompt), imageUrlMessage(dataUrl)], + ...(signal ? { signal } : {}), + }); + yield delta( + `${res.mime}, ${formatSize(res.bytes.length)} — described by ${describer.modelId}:\n`, + ); + let streamedAny = false; + let outcome: LLMOutcome | undefined; + for (;;) { + const step = await gen.next(); + if (step.done) { + outcome = step.value; + break; + } + const p = step.value.payload as { type?: string; event_type?: string; text?: string }; + if (p.type === "partial_text" && p.event_type === "delta" && p.text) { + streamedAny = true; + yield delta(p.text); + } + } + if (signal?.aborted) return { stopReason: "aborted" }; + if (!outcome || outcome.status !== "completed") { + const detail = + outcome && "message" in outcome && outcome.message ? `: ${outcome.message}` : ""; + yield delta( + `${streamedAny ? "\n" : ""}Vision model (${describer.modelId}) request ${outcome?.status ?? "failed"}${detail}`, + ); + return { stopReason: "failed" }; + } + if (!streamedAny) yield delta("[vision model returned no text]"); + }, + }; +} diff --git a/packages/core/src/environment/tools/exec-command.ts b/packages/core/src/environment/tools/exec-command.ts new file mode 100644 index 0000000..8f443f7 --- /dev/null +++ b/packages/core/src/environment/tools/exec-command.ts @@ -0,0 +1,121 @@ +/** + * exec_command —— local shell executor, a built-in tool implementation (BuiltinTool). + * + * Spawns a process inside the Workspace via `bash -lc ` and streams content deltas as + * stdout/stderr chunks arrive. Waits up to `yield_time_ms`: if the command finishes in time, + * returns the full output and exit status; if it's still running when the deadline hits, + * returns the output collected so far plus a `process_id` — the process moves to background, + * managed by `CommandSessionManager`, and is interacted with afterward via `input_command`. + * Completion is decided by **the foreground + * process exiting**, not by waiting for EOF on the output stream — background children (e.g. + * `node server.js &`) that inherit the pipes won't hold up the tool. + * + * Division of responsibility with Environment (see environment.ts): this tool only produces + * content deltas; exit code/signal/spawn errors and `process_id` are reported via the return + * value's `note` (appended outside the maxOutputLength truncation, so it survives even when + * long output gets truncated). Whether it ends normally or abnormally, it always finishes via + * the return value, **never throws**; on interruption it only reports `aborted` — the + * interruption note itself is appended by Environment. + * Docs: /docs/tools § "Command sessions". + */ +import path from "node:path"; +import { partialToolCallOutput } from "../../omnimessage/index.js"; +import type { OmniMessage } from "../../omnimessage/index.js"; +import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js"; +import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js"; +import { DEFAULT_EXEC_YIELD_MS, resultForExit } from "./command/index.js"; +import { clampYield } from "./background/index.js"; + +/** Tool name constant (used only inside this tool module, not exposed to Environment). */ +export const EXEC_COMMAND_NAME = "exec_command"; + +/** + * exec_command built-in tool: parses arguments, resolves workdir, and delegates to + * `CommandSessionManager` to spawn the process and collect output. + * `definition` is overridden by Environment at construction time with the matching entry + * from ToolConfig (description/parameters/permission/limits). + * `services.commandSessions` is injected by Environment (shares the same registry with + * input_command). + */ +export function createExecCommandTool( + definition: ToolDefinitionConfig, + services?: EnvironmentServices, +): BuiltinTool { + const manager = services?.commandSessions; + return { + name: EXEC_COMMAND_NAME, + definition, + async *execute( + args: Record, + ctx: ToolExecutionContext, + ): AsyncGenerator { + const { toolCallId, signal } = ctx; + const delta = (output: string): OmniMessage => + partialToolCallOutput({ eventType: "delta", output, toolCallId }); + + if (!manager) { + yield delta("[exec_command unavailable: no command session manager configured]"); + return { stopReason: "failed" }; + } + + const cmd = args["cmd"]; + if (typeof cmd !== "string" || cmd.length === 0) { + yield delta('Missing required argument "cmd" for exec_command.'); + return { stopReason: "failed" }; + } + // workdir defaults to workspaceDir; relative paths are resolved against workspaceDir. + const rawWorkdir = args["workdir"]; + const workdir = + typeof rawWorkdir === "string" && rawWorkdir.length > 0 + ? path.resolve(ctx.workspaceDir, rawWorkdir) + : ctx.workspaceDir; + const yieldMs = clampYield( + args["yield_time_ms"], + DEFAULT_EXEC_YIELD_MS, + definition.timeoutMs, + ); + + // Caller already aborted: finish immediately with aborted. + if (signal?.aborted) return { stopReason: "aborted" }; + + let session; + try { + session = manager.spawn({ cmd, cwd: workdir }); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + yield delta(`[spawn error: ${message}]`); + return { stopReason: "failed" }; + } + + // On interruption, kill the whole process group (background children included) to + // avoid orphans; once the process moves to background this listener is removed in finally. + const onAbort = (): void => session.kill(); + let registered = false; + signal?.addEventListener("abort", onAbort, { once: true }); + try { + for await (const chunk of session.collect(yieldMs, signal)) yield delta(chunk); + + if (signal?.aborted) return { stopReason: "aborted" }; + if (session.running) { + // Still running at the deadline: register as a background process, returning + // process_id so input_command can continue accessing it. + const id = manager.register(session); + registered = true; + return { + stopReason: "completed", + note: `[process running with process_id ${id}; use input_command to send input or poll for output]`, + }; + } + // Already exited: report exit status; process group cleanup (reaping any leftover + // background children) is handled uniformly in finally. + if (session.error) { + return { stopReason: "failed", note: `[spawn error: ${session.error.message}]` }; + } + return resultForExit(session.exit); + } finally { + signal?.removeEventListener("abort", onAbort); + if (!registered) session.kill(); + } + }, + }; +} diff --git a/packages/core/src/environment/tools/input-command.ts b/packages/core/src/environment/tools/input-command.ts new file mode 100644 index 0000000..8f5e3b5 --- /dev/null +++ b/packages/core/src/environment/tools/input-command.ts @@ -0,0 +1,109 @@ +/** + * input_command — accesses a long-running command session started by `exec_command` (BuiltinTool). + * + * Finds the session by `process_id`: if `chars` is non-empty, writes it to stdin first (when it is exactly `\u0003`, special-cased as sending SIGINT to the + * process group, i.e. Ctrl-C — it must be sent alone; mixing it with other content errors out, + * since a pipe has no terminal line discipline and a mixed-in ETX byte would just be written + * into stdin silently with no effect), and if empty, nothing is written and it only polls. + * It then collects new output within `yield_time_ms` or waits for exit. If the command is still + * running, returns the same `process_id`; once exited, returns the trailing output and exit + * status and cleans up the session. + * + * Shares the same `CommandSessionManager` injected by Environment with exec_command. An + * interruption only cancels this poll — **it does not kill the background process** (the process + * was started independently earlier; interrupting one poll shouldn't kill it as a side effect). + * Docs: /docs/tools § "Command sessions". + */ +import { partialToolCallOutput } from "../../omnimessage/index.js"; +import type { OmniMessage } from "../../omnimessage/index.js"; +import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js"; +import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js"; +import { + DEFAULT_EMPTY_POLL_YIELD_MS, + DEFAULT_WRITE_YIELD_MS, + resultForExit, +} from "./command/index.js"; +import { clampYield } from "./background/index.js"; + +/** Tool name constant. */ +export const INPUT_COMMAND_NAME = "input_command"; + +/** Ctrl-C: the ETX control character (U+0003). Received alone, it sends SIGINT to the process group instead of writing the byte into stdin. */ +const INTERRUPT = String.fromCharCode(3); // U+0003 (ETX) + +export function createInputCommandTool( + definition: ToolDefinitionConfig, + services?: EnvironmentServices, +): BuiltinTool { + const manager = services?.commandSessions; + return { + name: INPUT_COMMAND_NAME, + definition, + async *execute( + args: Record, + ctx: ToolExecutionContext, + ): AsyncGenerator { + const { toolCallId, signal } = ctx; + const delta = (output: string): OmniMessage => + partialToolCallOutput({ eventType: "delta", output, toolCallId }); + + if (!manager) { + yield delta("[input_command unavailable: no command session manager configured]"); + return { stopReason: "failed" }; + } + + const processId = args["process_id"]; + if (typeof processId !== "string" || processId.length === 0) { + yield delta('Missing required argument "process_id" for input_command.'); + return { stopReason: "failed" }; + } + const session = manager.get(processId); + if (!session) { + yield delta( + `[input_command error: unknown process_id ${processId} (the session may have exited and been cleared)]`, + ); + return { stopReason: "failed" }; + } + + const chars = typeof args["chars"] === "string" ? (args["chars"] as string) : ""; + const empty = chars.length === 0; + const yieldMs = clampYield( + args["yield_time_ms"], + empty ? DEFAULT_EMPTY_POLL_YIELD_MS : DEFAULT_WRITE_YIELD_MS, + definition.timeoutMs, + ); + + if (signal?.aborted) return { stopReason: "aborted" }; + + // Write / interrupt (empty chars just polls). U+0003 mixed with other content errors out + // rather than being written silently (same as codex): a pipe has no terminal line + // discipline, so an ETX byte in stdin produces no interruption — the model would just + // see the command still running. + if (!empty) { + if (chars === INTERRUPT) session.interrupt(); + else if (chars.includes(INTERRUPT)) { + yield delta( + '[input_command error: chars mixes U+0003 (Ctrl-C) with other content; send "\\u0003" alone to interrupt]', + ); + return { stopReason: "failed" }; + } else session.write(chars); + } + + for await (const chunk of session.collect(yieldMs, signal)) yield delta(chunk); + + if (signal?.aborted) return { stopReason: "aborted" }; + if (session.running) { + return { + stopReason: "completed", + note: `[process still running with process_id ${processId}]`, + }; + } + // Already exited: clean up the registry and report the exit status. + manager.remove(processId); + if (session.error) { + return { stopReason: "failed", note: `[spawn error: ${session.error.message}]` }; + } + return resultForExit(session.exit); + }, + }; +} diff --git a/packages/core/src/environment/tools/input-subagent.ts b/packages/core/src/environment/tools/input-subagent.ts new file mode 100644 index 0000000..b7f84e7 --- /dev/null +++ b/packages/core/src/environment/tools/input-subagent.ts @@ -0,0 +1,125 @@ +/** + * input_subagent —— accesses a subagent session that `run_subagent` moved to the background + * (BuiltinTool). + * + * Finds the session by `subagent_id`: when `prompt` is empty, nothing is written — it just + * polls (collecting subagent messages and text deltas buffered during the background period, + * or waiting for the run to end); when non-empty and the subagent is idle, it's fed in as a new + * user message to continue on the same child Session (long-running subagent, multi-turn + * conversation); when non-empty but the subagent is still running, it errors, suggesting to + * poll first. Within the window, queued approval requests from the child session are also + * passed through (see subagent/session.ts). + * + * Difference from `input_command`: once a round of work finishes, the session is **not + * removed** (kept to receive a follow-up prompt) — it's only released when the parent Session + * ends, or evicted as an idle session once concurrency is full. Interruption only aborts this + * particular access, **it never kills the child session** (the subagent was launched + * independently earlier; the user interrupting one poll shouldn't kill it along the way). + * Docs: /docs/tools § "Subagents". + */ +import { partialToolCallOutput } from "../../omnimessage/index.js"; +import type { OmniMessage } from "../../omnimessage/index.js"; +import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js"; +import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js"; +import { + DEFAULT_SUBAGENT_POLL_YIELD_MS, + DEFAULT_SUBAGENT_YIELD_MS, + resultForSubagentExit, +} from "./subagent/index.js"; +import { approvalHint } from "./run-subagent.js"; +import { collectWindow } from "./subagent/collect.js"; +import { clampYield } from "./background/index.js"; + +/** Tool name constant. */ +export const INPUT_SUBAGENT_NAME = "input_subagent"; + +export function createInputSubagentTool( + definition: ToolDefinitionConfig, + services?: EnvironmentServices, +): BuiltinTool { + const manager = services?.subagentSessions; + return { + name: INPUT_SUBAGENT_NAME, + definition, + async *execute( + args: Record, + ctx: ToolExecutionContext, + ): AsyncGenerator { + const { toolCallId, signal, approve } = ctx; + const delta = (output: string): OmniMessage => + partialToolCallOutput({ eventType: "delta", output, toolCallId }); + + if (!manager) { + yield delta("[input_subagent unavailable: no subagent session manager configured]"); + return { stopReason: "failed" }; + } + + const subagentId = args["subagent_id"]; + if (typeof subagentId !== "string" || subagentId.length === 0) { + yield delta('Missing required argument "subagent_id" for input_subagent.'); + return { stopReason: "failed" }; + } + const session = manager.get(subagentId); + if (!session) { + yield delta( + `[input_subagent error: unknown subagent_id ${subagentId} ` + + `(the session may have finished and been cleared)]`, + ); + return { stopReason: "failed" }; + } + + const prompt = typeof args["prompt"] === "string" ? (args["prompt"] as string) : ""; + const empty = prompt.trim().length === 0; + const yieldMs = clampYield( + args["yield_time_ms"], + empty ? DEFAULT_SUBAGENT_POLL_YIELD_MS : DEFAULT_SUBAGENT_YIELD_MS, + definition.timeoutMs, + ); + + if (signal?.aborted) return { stopReason: "aborted" }; + + // Continue with a follow-up prompt (empty prompt just polls). New input is not accepted + // while running: poll first to collect progress. + if (!empty) { + if (session.running) { + yield delta( + `[input_subagent error: subagent ${subagentId} is still running; ` + + `poll with an empty prompt to collect progress first]`, + ); + return { stopReason: "failed" }; + } + // startRun expresses edge cases like already-disposed via throw, collapsed here into failed (the tool never throws outward). + try { + session.startRun(prompt); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + yield delta(`[input_subagent error: ${message}]`); + return { stopReason: "failed" }; + } + } + + yield* collectWindow(session, { + yieldMs, + toolCallId, + ...(signal ? { signal } : {}), + ...(approve ? { approve } : {}), + }); + + // Interruption only aborts this access, it doesn't kill the child session. + if (signal?.aborted) return { stopReason: "aborted" }; + if (session.running) { + return { + stopReason: "completed", + note: `[subagent still running with subagent_id ${subagentId}]` + approvalHint(session), + }; + } + // This round of work has ended: report the end state; the session is kept (can be resumed), not removed from the registry. + const result = resultForSubagentExit(session.exit); + const idleHint = `[subagent idle with subagent_id ${subagentId}; send a follow-up prompt to continue]`; + return { + ...result, + note: result.note !== undefined ? `${result.note} ${idleHint}` : idleHint, + }; + }, + }; +} diff --git a/packages/core/src/environment/tools/read-image.ts b/packages/core/src/environment/tools/read-image.ts new file mode 100644 index 0000000..ef63504 --- /dev/null +++ b/packages/core/src/environment/tools/read-image.ts @@ -0,0 +1,248 @@ +/** + * read_image — image-reading tool, a builtin tool implementation (BuiltinTool). + * + * Reads an image and feeds it back to the model as **image content**: if `source` is an http(s) + * URL, downloads it with the global fetch (respecting the abort signal); otherwise reads it as a + * local file path (relative paths are resolved against the Workspace). Only png/jpeg/gif/webp + * are allowed (determined in order by response header / magic number / extension); errors out + * above 5MB. + * + * Division of responsibility with Environment (see environment.ts): on success, yields a brief + * descriptive delta (e.g. `image/png, 123.4 kB`), while the image itself is carried via the + * return value `ToolResult.images` (a data URL) for Environment to attach when closing out (a + * single streaming delta carries it all at once before stop, plus the final complete + * `tool_call_output`); on failure, yields explanatory text and closes with `failed`, **never + * throwing**; if interrupted, only reports `aborted` — the interruption note is appended by + * Environment. + * + * This tool is only used by sessions with a model that supports images (config entry + * `forModel: "vision"`); text-only models use describe_image instead (the image is handed to a + * configured vision model to describe, returning text — see describe-image.ts), and the image + * loading/validation logic is shared via `loadImage`. + * Docs: /docs/tools § "Image tools". + */ +import path from "node:path"; +import { readFile, stat } from "node:fs/promises"; +import { partialToolCallOutput } from "../../omnimessage/index.js"; +import type { OmniMessage } from "../../omnimessage/index.js"; +import type { ToolDefinitionConfig } from "../../interfaces.js"; +import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js"; + +/** Tool name constant (used only within this tool module, never exposed to Environment). */ +export const READ_IMAGE_NAME = "read_image"; + +/** + * Image size upper bound (bytes): errors out above this. Taken as the common denominator of + * per-provider single-image hard limits (Claude API is around 5MB, some compatible endpoints are + * lower) — since local validation passing but the next request getting a blanket 400 from the + * provider is a non-retryable path, the limit must not exceed the strictest downstream; this also + * avoids oversized images blowing up the context and Trace. + */ +export const MAX_IMAGE_BYTES = 5 * 1024 * 1024; + +/** Supported image mime types (the four generally accepted across providers). */ +const SUPPORTED_MIMES = new Set(["image/png", "image/jpeg", "image/gif", "image/webp"]); + +/** Extension -> mime (fallback when magic-number sniffing fails). */ +const EXT_TO_MIME: Record = { + ".png": "image/png", + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".gif": "image/gif", + ".webp": "image/webp", +}; + +/** Sniffs the mime type from the file header's magic number; returns null if unrecognized. */ +function sniffMime(buf: Buffer): string | null { + if (buf.length >= 8 && buf.readUInt32BE(0) === 0x89504e47) return "image/png"; + if (buf.length >= 3 && buf[0] === 0xff && buf[1] === 0xd8 && buf[2] === 0xff) { + return "image/jpeg"; + } + if (buf.length >= 6) { + const head = buf.subarray(0, 6).toString("latin1"); + if (head === "GIF87a" || head === "GIF89a") return "image/gif"; + } + if ( + buf.length >= 12 && + buf.subarray(0, 4).toString("latin1") === "RIFF" && + buf.subarray(8, 12).toString("latin1") === "WEBP" + ) { + return "image/webp"; + } + return null; +} + +/** Infers the mime type from a path / URL pathname's extension; returns null if it can't be inferred. */ +function mimeFromExt(p: string): string | null { + return EXT_TO_MIME[path.extname(p).toLowerCase()] ?? null; +} + +/** Byte count -> human-readable size (B / kB / MB, one decimal place). */ +export function formatSize(bytes: number): string { + if (bytes < 1024) return `${bytes} B`; + const kb = bytes / 1024; + if (kb < 1024) return `${kb.toFixed(1)} kB`; + return `${(kb / 1024).toFixed(1)} MB`; +} + +const OVERSIZE_MESSAGE = (size: number): string => + `Image too large: ${formatSize(size)} exceeds the ${formatSize(MAX_IMAGE_BYTES)} limit.`; + +const UNSUPPORTED_MESSAGE = (detected: string | null): string => + `Unsupported image type${detected ? ` "${detected}"` : ""}: only png, jpeg, gif and webp are supported.`; + +/** Result of `loadImage`: success (bytes + mime) / interrupted / failed (explanatory message). */ +export type LoadImageResult = + | { ok: true; bytes: Buffer; mime: string } + | { ok: false; reason: "aborted" } + | { ok: false; reason: "failed"; message: string }; + +/** + * Reads and validates an image (shared by read_image and describe_image): + * an http(s) URL is downloaded with the global fetch, otherwise read as a local path (resolved + * against Workspace); validates the size upper bound and mime type (determined in order by + * response header / magic number / extension). Never throws. + */ +export async function loadImage( + source: string, + workspaceDir: string, + signal?: AbortSignal, +): Promise { + if (signal?.aborted) return { ok: false, reason: "aborted" }; + + let bytes: Buffer; + let mime: string | null; + if (/^https?:\/\//i.test(source)) { + // URL branch: downloads via the global fetch (abort signal passed through to the request); + // mime is preferentially taken from the response header, falling back to magic number / URL + // extension. + let res: Response; + try { + res = await fetch(source, signal ? { signal } : {}); + } catch (err) { + if (signal?.aborted) return { ok: false, reason: "aborted" }; + const message = err instanceof Error ? err.message : String(err); + return { + ok: false, + reason: "failed", + message: `Failed to download image "${source}": ${message}`, + }; + } + if (!res.ok) { + return { + ok: false, + reason: "failed", + message: `Failed to download image "${source}": HTTP ${res.status}`, + }; + } + // When content-length is trustworthy, reject an oversized response early to avoid reading it + // into memory for nothing. + const declared = Number(res.headers.get("content-length") ?? ""); + if (Number.isFinite(declared) && declared > MAX_IMAGE_BYTES) { + return { ok: false, reason: "failed", message: OVERSIZE_MESSAGE(declared) }; + } + try { + bytes = Buffer.from(await res.arrayBuffer()); + } catch (err) { + if (signal?.aborted) return { ok: false, reason: "aborted" }; + const message = err instanceof Error ? err.message : String(err); + return { + ok: false, + reason: "failed", + message: `Failed to download image "${source}": ${message}`, + }; + } + const headerMime = (res.headers.get("content-type") ?? "").split(";")[0]!.trim().toLowerCase(); + let urlExtMime: string | null = null; + try { + urlExtMime = mimeFromExt(new URL(source).pathname); + } catch { + urlExtMime = null; // A URL parse failure only affects the extension fallback + } + mime = SUPPORTED_MIMES.has(headerMime) ? headerMime : (sniffMime(bytes) ?? urlExtMime); + if (mime === null && headerMime) mime = headerMime; // Include the real response type in the error + } else { + // Local-path branch: relative paths are resolved against Workspace; stat first to check the + // size before reading, to avoid reading an oversized file into memory in one go. + const filePath = path.resolve(workspaceDir, source); + try { + const st = await stat(filePath); + // Explicitly reject non-file paths such as directories: readFile's EISDIR error isn't + // model-friendly. + if (!st.isFile()) { + return { + ok: false, + reason: "failed", + message: `Failed to read image "${source}": path is not a file.`, + }; + } + if (st.size > MAX_IMAGE_BYTES) { + return { ok: false, reason: "failed", message: OVERSIZE_MESSAGE(st.size) }; + } + bytes = await readFile(filePath); + } catch (err) { + if (signal?.aborted) return { ok: false, reason: "aborted" }; + const message = err instanceof Error ? err.message : String(err); + return { + ok: false, + reason: "failed", + message: `Failed to read image "${source}": ${message}`, + }; + } + mime = sniffMime(bytes) ?? mimeFromExt(filePath); + } + + if (signal?.aborted) return { ok: false, reason: "aborted" }; + // Empty file/response: magic-number sniffing fails to identify it, but the extension fallback + // may still let it through — an empty base64 sent to the provider is guaranteed to error, so + // reject it here. + if (bytes.length === 0) { + return { ok: false, reason: "failed", message: `Image "${source}" is empty.` }; + } + if (bytes.length > MAX_IMAGE_BYTES) { + return { ok: false, reason: "failed", message: OVERSIZE_MESSAGE(bytes.length) }; + } + if (mime === null || !SUPPORTED_MIMES.has(mime)) { + return { ok: false, reason: "failed", message: UNSUPPORTED_MESSAGE(mime) }; + } + return { ok: true, bytes, mime }; +} + +/** + * read_image builtin tool: reads a local file or downloads a URL, validates its type and size, + * then outputs a data URL image. `definition` is overridden by Environment at construction time + * with the same-named entry from ToolConfig (description/arguments/permissions/limits). + */ +export function createReadImageTool(definition: ToolDefinitionConfig): BuiltinTool { + return { + name: READ_IMAGE_NAME, + definition, + async *execute( + args: Record, + ctx: ToolExecutionContext, + ): AsyncGenerator { + const { toolCallId, signal } = ctx; + const delta = (output: string): OmniMessage => + partialToolCallOutput({ eventType: "delta", output, toolCallId }); + + const source = args["source"]; + if (typeof source !== "string" || source.length === 0) { + yield delta('Missing required argument "source" for read_image.'); + return { stopReason: "failed" }; + } + + const res = await loadImage(source, ctx.workspaceDir, signal); + if (!res.ok) { + if (res.reason === "aborted") return { stopReason: "aborted" }; + yield delta(res.message); + return { stopReason: "failed" }; + } + + // Success: yield a brief one-line description as a text delta (both in the streaming and + // complete message), while the image itself is carried via the return value for + // Environment to attach. + yield delta(`${res.mime}, ${formatSize(res.bytes.length)}`); + return { images: [`data:${res.mime};base64,${res.bytes.toString("base64")}`] }; + }, + }; +} diff --git a/packages/core/src/environment/tools/registry.ts b/packages/core/src/environment/tools/registry.ts new file mode 100644 index 0000000..a9855b5 --- /dev/null +++ b/packages/core/src/environment/tools/registry.ts @@ -0,0 +1,47 @@ +/** + * Built-in tool registry —— maps tool names to BuiltinTool factories. + * + * Environment uses this table to assemble entries from ToolConfig into BuiltinTool instances: + * a tool is only assembled if its name is in the table (i.e. a supported built-in tool); the + * description/parameters/permission/maxOutputLength from config are injected into the tool's + * `definition` by each factory. + * When adding a new built-in tool, just register one factory entry here — no changes to + * Environment needed. + * + * Docs: packages/docs/content/tools.{zh,en}.md (site path /docs/tools) documents every + * built-in tool and the approval flow — keep the page in sync when this table changes. + */ +import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js"; +import type { BuiltinTool } from "./types.js"; +import { EXEC_COMMAND_NAME, createExecCommandTool } from "./exec-command.js"; +import { INPUT_COMMAND_NAME, createInputCommandTool } from "./input-command.js"; +import { SUBAGENT_NAME, createSubagentTool } from "./run-subagent.js"; +import { INPUT_SUBAGENT_NAME, createInputSubagentTool } from "./input-subagent.js"; +import { READ_IMAGE_NAME, createReadImageTool } from "./read-image.js"; +import { DESCRIBE_IMAGE_NAME, createDescribeImageTool } from "./describe-image.js"; + +/** + * A factory that constructs a BuiltinTool instance from a tool config entry; optionally + * receives runtime services injected by Environment. + * Most tools ignore `services`; only a few (e.g. `run_subagent`) use it. + */ +export type BuiltinToolFactory = ( + definition: ToolDefinitionConfig, + services?: EnvironmentServices, +) => BuiltinTool; + +/** Tool name -> factory. */ +export const BUILTIN_TOOL_FACTORIES: Record = { + [EXEC_COMMAND_NAME]: createExecCommandTool, + [INPUT_COMMAND_NAME]: createInputCommandTool, + [SUBAGENT_NAME]: createSubagentTool, + [INPUT_SUBAGENT_NAME]: createInputSubagentTool, + [READ_IMAGE_NAME]: createReadImageTool, + // describe_image: the text-only-model variant of read_image (hands the image to the + // configured vision model for description, returns text). + // Which tool is used for which model class is declared by the config entry's forModel + // annotation; before assembly, selectBuiltinToolsForModel has already filtered out entries + // that don't apply to the session's model. + [DESCRIBE_IMAGE_NAME]: (definition, services) => + createDescribeImageTool(definition, services?.visionDescriber ?? { modelId: null }), +}; diff --git a/packages/core/src/environment/tools/run-subagent.ts b/packages/core/src/environment/tools/run-subagent.ts new file mode 100644 index 0000000..9c27cf9 --- /dev/null +++ b/packages/core/src/environment/tools/run-subagent.ts @@ -0,0 +1,154 @@ +/** + * run_subagent — delegates a subtask to a child Agent, supporting a switch to long-running + * background execution. + * + * The tool itself doesn't depend on Agent/Session, only holding an injected `SubagentRunner` + * (breaking the circular dependency). The model may freely choose the child Agent (`agent_id`) + * and model (`model_id`) via arguments; if omitted, it falls back to reusing the current Agent + * and the Project's default Model respectively. The spawned child session is managed by + * `ManagedSubagentSession` (sharing the `SubagentSessionManager` injected by Environment with + * `input_subagent`). + * + * The two-phase semantics mirror `exec_command`: within the `yield_time_ms` window, child-session + * messages are forwarded live (tagged with origin, so the frontend can see the child Agent's tool + * calls and token usage) and the child Agent's text deltas are copied as this tool's output; if + * the child Agent finishes within the window, its terminal state is returned and the child + * session is released; if it's still running once the window expires, it's registered as a + * background session, returning `subagent_id` for subsequent access the same way as + * `input_command` (polling / appending a Prompt, see input-subagent.ts). + * + * Approval: `run_subagent` itself is a read-write tool (`rw`), so its invocation requires Human + * approval; the child session's tool approval requests are forwarded to the same Human within + * the window via the session's approval queue (tagged with origin), and queued for the next + * access while running in the background. An interruption within the startup window kills the + * child session per exec_command semantics; precheck errors such as exceeding the depth limit or + * a nonexistent agent are expressed by the runner as a throw, and collapsed to failed. + * Docs: /docs/tools § "Subagents". + */ +import { partialToolCallOutput } from "../../omnimessage/index.js"; +import type { OmniMessage } from "../../omnimessage/index.js"; +import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js"; +import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js"; +import { + DEFAULT_SUBAGENT_YIELD_MS, + ManagedSubagentSession, + resultForSubagentExit, +} from "./subagent/index.js"; +import { collectWindow } from "./subagent/collect.js"; +import { clampYield } from "./background/index.js"; + +/** Tool name constant (used only within this tool module, never exposed to Environment). */ +export const SUBAGENT_NAME = "run_subagent"; + +/** Pending-approval hint: lets the model know it should poll again to move the child Agent forward. */ +export function approvalHint(session: ManagedSubagentSession): string { + const n = session.pendingApprovals; + return n > 0 ? ` [subagent is waiting for approval of ${n} tool call(s); poll to review]` : ""; +} + +/** Builds run_subagent's BuiltinTool from tool config + injected services. */ +export function createSubagentTool( + definition: ToolDefinitionConfig, + services?: EnvironmentServices, +): BuiltinTool { + const runner = services?.subagentRunner; + const manager = services?.subagentSessions; + return { + name: SUBAGENT_NAME, + definition, + async *execute( + args: Record, + ctx: ToolExecutionContext, + ): AsyncGenerator { + const { toolCallId, signal, approve } = ctx; + const fail = function* (msg: string): Generator { + yield partialToolCallOutput({ eventType: "delta", output: msg, toolCallId }); + }; + + // Missing arguments / unconfigured services both collapse to an explanatory output rather + // than throwing (consistent with other tools). + if (!runner) { + yield* fail("[run_subagent unavailable: no subagent runner configured]"); + return { stopReason: "failed" }; + } + if (!manager || manager.isDisposed) { + yield* fail("[run_subagent unavailable: no subagent session manager available]"); + return { stopReason: "failed" }; + } + const prompt = typeof args.prompt === "string" ? args.prompt : ""; + if (prompt.trim().length === 0) { + yield* fail("[run_subagent error: missing required string argument `prompt`]"); + return { stopReason: "failed" }; + } + const agentId = typeof args.agent_id === "string" ? args.agent_id : undefined; + const modelId = typeof args.model_id === "string" ? args.model_id : undefined; + const yieldMs = clampYield( + args.yield_time_ms, + DEFAULT_SUBAGENT_YIELD_MS, + definition.timeoutMs, + ); + + if (signal?.aborted) return { stopReason: "aborted" }; + + // Concurrency cap (running child Agents are never evicted): reject spawning if there's no + // room. + if (!manager.makeRoom()) { + yield* fail( + "[run_subagent error: too many background subagents; poll or finish existing ones first]", + ); + return { stopReason: "failed" }; + } + + // Spawn the child Session (precheck errors such as exceeding the depth limit or a + // nonexistent agent are expressed as a throw). + let session: ManagedSubagentSession; + try { + const handle = await runner.spawn({ + ...(agentId !== undefined ? { agentId } : {}), + ...(modelId !== undefined ? { modelId } : {}), + }); + session = new ManagedSubagentSession(handle); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + yield* fail(`[run_subagent error: ${message}]`); + return { stopReason: "failed" }; + } + + // An interruption within the startup window kills the child session (consistent with + // exec_command); once switched to background, this listener is removed in `finally`. + const onAbort = (): void => session.kill(); + let registered = false; + signal?.addEventListener("abort", onAbort, { once: true }); + try { + session.startRun(prompt); + yield* collectWindow(session, { + yieldMs, + toolCallId, + ...(signal ? { signal } : {}), + ...(approve ? { approve } : {}), + }); + + if (signal?.aborted) return { stopReason: "aborted" }; + if (session.running) { + // Still running once the window expires: register as a background session, returning + // subagent_id for input_subagent to continue accessing it. + const id = manager.register(session); + registered = true; + return { + stopReason: "completed", + note: + `[subagent running with subagent_id ${id}; use input_subagent to poll for progress ` + + `or send a follow-up prompt]` + + approvalHint(session), + }; + } + // Finished within the window: report the terminal state; releasing the child session is + // handled uniformly in `finally` (never registered, so no subagent_id). + return resultForSubagentExit(session.exit); + } finally { + signal?.removeEventListener("abort", onAbort); + if (!registered) session.kill(); + } + }, + }; +} diff --git a/packages/core/src/environment/tools/subagent/collect.ts b/packages/core/src/environment/tools/subagent/collect.ts new file mode 100644 index 0000000..e316e0a --- /dev/null +++ b/packages/core/src/environment/tools/subagent/collect.ts @@ -0,0 +1,52 @@ +/** + * collectWindow —— yield-window collector shared by run_subagent / input_subagent. + * + * Within the `yieldMs` window, emits child-session output in real time: buffered child-session + * messages (already origin-tagged, passed through to the frontend) and subagent text deltas + * (fed back to the LLM as this parent tool's own output delta). The window also hooks up an + * approval outlet, forwarding the child session's queued approval requests one by one to the + * Human via `approve`. The window ends on "run finished / signal abort / deadline reached", and + * does a final drain right before ending (to catch the tail buffer at the moment the run + * finishes). Deciding the end state and finalizing are the caller's responsibility. + */ +import { partialToolCallOutput } from "../../../omnimessage/index.js"; +import type { OmniMessage } from "../../../omnimessage/index.js"; +import type { ApproveFn } from "../../../interfaces.js"; +import type { ManagedSubagentSession } from "./session.js"; + +export async function* collectWindow( + session: ManagedSubagentSession, + opts: { yieldMs: number; toolCallId: string; signal?: AbortSignal; approve?: ApproveFn }, +): AsyncGenerator { + const { yieldMs, toolCallId, signal, approve } = opts; + const delta = (output: string): OmniMessage => + partialToolCallOutput({ eventType: "delta", output, toolCallId }); + const detach = approve ? session.attachApprovalSink(approve) : null; + // abort only ends this window (whether to kill the child session is up to the caller); + // wakes up a pending waitWake so it returns immediately. + const onAbort = (): void => session.wakeup(); + signal?.addEventListener("abort", onAbort, { once: true }); + try { + const start = Date.now(); + for (;;) { + for (const m of session.drainMessages()) yield m; + const text = session.drainText(); + if (text) yield delta(text); + if (!session.running) break; + if (signal?.aborted) break; + const remaining = yieldMs - (Date.now() - start); + if (remaining <= 0) break; + // Re-check the predicate before sleeping: output arriving while `yield` is suspended + // would fire its wakeup before this wait even starts, and get missed otherwise. + if (session.hasPending) continue; + await session.waitWake(remaining); + } + // Final drain: there may still be a tail buffer right when the run finishes/yields. + for (const m of session.drainMessages()) yield m; + const tail = session.drainText(); + if (tail) yield delta(tail); + } finally { + signal?.removeEventListener("abort", onAbort); + detach?.(); + } +} diff --git a/packages/core/src/environment/tools/subagent/index.ts b/packages/core/src/environment/tools/subagent/index.ts new file mode 100644 index 0000000..840b7a1 --- /dev/null +++ b/packages/core/src/environment/tools/subagent/index.ts @@ -0,0 +1,7 @@ +/** + * Barrel for the background subagent session module. + */ +export { SubagentSessionManager } from "./session-manager.js"; +export { ManagedSubagentSession, resultForSubagentExit } from "./session.js"; +export type { SubagentExit } from "./session.js"; +export { DEFAULT_SUBAGENT_YIELD_MS, DEFAULT_SUBAGENT_POLL_YIELD_MS } from "./limits.js"; diff --git a/packages/core/src/environment/tools/subagent/limits.ts b/packages/core/src/environment/tools/subagent/limits.ts new file mode 100644 index 0000000..9afd20a --- /dev/null +++ b/packages/core/src/environment/tools/subagent/limits.ts @@ -0,0 +1,11 @@ +/** + * Default yield duration for background subagent sessions. + * + * See `../background/limits.ts` for the clamping logic: only a lower bound is set, and the + * upper bound is derived from the tool's own `timeoutMs`. + */ + +/** Default wait duration (ms) for `run_subagent` launching a task and `input_subagent` appending a Prompt to continue. */ +export const DEFAULT_SUBAGENT_YIELD_MS = 300_000; +/** Default wait duration (ms) for `input_subagent` empty polling. */ +export const DEFAULT_SUBAGENT_POLL_YIELD_MS = 10_000; diff --git a/packages/core/src/environment/tools/subagent/session-manager.ts b/packages/core/src/environment/tools/subagent/session-manager.ts new file mode 100644 index 0000000..8da6630 --- /dev/null +++ b/packages/core/src/environment/tools/subagent/session-manager.ts @@ -0,0 +1,63 @@ +/** + * SubagentSessionManager —— registry and lifecycle management for background subagent sessions. + * + * Constructed by Environment (one per Session), injected via services to be shared by the + * `run_subagent` and `input_subagent` tools. Registry duties are handled by the generic + * `BackgroundRegistry` (shared with command sessions, see `../background/registry.ts`). + * Difference from command sessions: when at capacity, **running sessions are never evicted** + * (discarding in-progress subagent work is unacceptable) — only completed, idle ones are + * evicted; if there's still no room, the tool rejects spawning a new one. + * Docs: /docs/tools § "Background session caps". + */ +import { BackgroundRegistry } from "../background/index.js"; +import type { ManagedSubagentSession } from "./session.js"; + +/** + * Cap on concurrently managed background subagent sessions. This is a **spawn admission cap**, + * not a hard limit: there's an await between the `makeRoom` check (before spawn) and `register` + * (after the yield window ends), so parallel run_subagent calls can briefly push the registered + * count over the cap — an already-running child session is never discarded just to hold the line. + */ +const MAX_SESSIONS = 8; + +export class SubagentSessionManager { + private readonly registry = new BackgroundRegistry({ + idPrefix: "subagent", + maxTasks: MAX_SESSIONS, + }); + + /** Whether the manager has been disposed (the host Session has ended). */ + get isDisposed(): boolean { + return this.registry.isDisposed; + } + + /** Whether there's still room for a new background session (evicting a completed, idle one if needed; never evicts a running one). */ + makeRoom(): boolean { + return this.registry.makeRoom(false); + } + + /** + * Registers a still-running session as a background session, allocating and returning a + * unique `subagent_id`: `subagent-` (falls back to random on + * collision), whose suffix aligns with the message origin/frontend nesting label + * (`agent-`) for correlation. + */ + register(session: ManagedSubagentSession): string { + // A full yield window has elapsed since the pre-spawn makeRoom check, so the registry may + // have been filled by parallel calls in the meantime: free up room once more (only evicting + // completed, idle ones); if still no room, register anyway, tolerating a brief overshoot + // (see MAX_SESSIONS). + this.registry.makeRoom(false); + return this.registry.register(session, session.sessionId.slice(-8)); + } + + /** Looks up a session by subagent_id and refreshes its access time; returns undefined if not found. */ + get(subagentId: string): ManagedSubagentSession | undefined { + return this.registry.get(subagentId); + } + + /** Disposes: removes the fallback registration and finalizes all sessions (the process 'exit' fallback is hooked by the registry itself). Idempotent. */ + dispose(): void { + this.registry.dispose(); + } +} diff --git a/packages/core/src/environment/tools/subagent/session.ts b/packages/core/src/environment/tools/subagent/session.ts new file mode 100644 index 0000000..7c0e7dc --- /dev/null +++ b/packages/core/src/environment/tools/subagent/session.ts @@ -0,0 +1,304 @@ +/** + * ManagedSubagentSession — a subagent session capable of running in the background. + * + * Holds a `SubagentHandle` and drives its `run` (pump): the same child Session may run across + * multiple rounds (the first round is initiated by `run_subagent`, later rounds append a Prompt + * via `input_subagent`). Structurally mirrors a command session (ManagedSession): the parent tool + * call collects output live within the yield window; output produced outside the window (while + * running in the background) goes into a buffer, delivered all at once on the next access. + * + * Three kinds of output, each with its own destination: + * - **Message buffer**: all of the child session's OmniMessage (already tagged with origin), for + * the parent tool call to forward to the frontend for rendering; capped in count, overflow + * drops the oldest (only affects frontend replay — the child Session's own Trace loses no + * data); + * - **Text buffer**: assistant text deltas from the direct child layer (origin one hop), fed back + * as the parent tool's own output to the LLM; capped in capacity (prevents memory bloat), + * overflow drops the oldest with a marker; + * - **Approval queue**: the child session's tool approval requests. While running in the + * background, the parent session may have no active tool call to forward approval through, so + * the request is queued and the child session blocks waiting; the parent tool call + * (run_subagent / input_subagent) attaches an approval sink (`attachApprovalSink`) within its + * window to consult Human one request at a time — if detached mid-consultation (window ends), + * the request stays queued, and a late-arriving decision still takes effect (settled guard, + * first to arrive wins). + * + * Cleanup: `kill()` aborts the current run via AbortSignal, denies all pending approvals, and + * releases child Session resources; idempotent. The child Session runs in-process, so there's no + * need for a separate synchronous hard-kill path (its command sessions are reaped by their own + * exit fallback); `killHard` is equivalent to `kill`. + */ +import type { OmniMessage } from "../../../omnimessage/index.js"; +import type { ApprovalDecision, ToolCallPayload } from "../../../omnimessage/index.js"; +import type { ApproveFn, SubagentHandle } from "../../../interfaces.js"; +import type { ToolResult } from "../types.js"; +import { CappedTextBuffer, WakeSignal } from "../background/index.js"; + +/** Message buffer count cap: overflow drops the oldest (only frontend replay is affected — the child Session has its own Trace). */ +const MESSAGE_BUFFER_CAP = 4096; +/** Text buffer capacity cap (characters): prevents a chatty child Agent from blowing up memory. */ +const OUTPUT_BUFFER_CAP = 1024 * 1024; // 1 MiB + +/** Terminal state of one run. */ +export interface SubagentExit { + status: "completed" | "failed"; + note?: string; +} + +/** A pending approval request: the settled guard makes the decision first-to-arrive-wins (late/duplicate decisions are ignored). */ +interface PendingApproval { + toolCall: OmniMessage; + settled: boolean; + resolve: (decision: ApprovalDecision) => void; +} + +export class ManagedSubagentSession { + /** Timestamp of the last access (used for the eviction policy). */ + lastUsed: number = Date.now(); + + private readonly handle: SubagentHandle; + private readonly abortCtrl = new AbortController(); + + private messages: OmniMessage[] = []; + private readonly textBuffer = new CappedTextBuffer(OUTPUT_BUFFER_CAP, "earlier subagent output"); + + private isRunning = false; + private exitInfo: SubagentExit | null = null; + private killed = false; + + private readonly approvals: PendingApproval[] = []; + private sink: { approve: ApproveFn; detached: Promise } | null = null; + private sinkEpoch = 0; + private pumpingApprovals = false; + + // Single wake point: new message / run finished / new approval request all wake a waiting waitWake through it. + private readonly wakeSignal = new WakeSignal(); + + constructor(handle: SubagentHandle) { + this.handle = handle; + } + + /** Child Session id (one hop of a message's origin); `subagent_id` is derived from its tail so the frontend can correlate it. */ + get sessionId(): string { + return this.handle.sessionId; + } + + /** Whether a round of the task is currently running. */ + get running(): boolean { + return this.isRunning; + } + + /** Terminal state of the most recent run; null if no round has ever completed. */ + get exit(): SubagentExit | null { + return this.exitInfo; + } + + /** Number of pending approval requests (the parent tool uses this to hint the model to poll again). */ + get pendingApprovals(): number { + return this.approvals.length; + } + + /** Whether there's unread output (buffered messages or text); used to re-check the predicate before waiting (see collect.ts). */ + get hasPending(): boolean { + return this.messages.length > 0 || !this.textBuffer.isEmpty; + } + + /** + * Starts a new round of the task on the child Session (async pump, doesn't block the caller). + * Throws if already disposed or still running (converted to an explanatory output by the + * caller). + */ + startRun(prompt: string): void { + if (this.killed) throw new Error("subagent session disposed"); + if (this.isRunning) throw new Error("subagent is still running"); + this.isRunning = true; + this.exitInfo = null; + void this.pump(prompt); + } + + /** Takes the buffered child-session messages (already tagged with origin, for the parent tool to forward). */ + drainMessages(): OmniMessage[] { + if (this.messages.length === 0) return []; + const out = this.messages; + this.messages = []; + return out; + } + + /** Takes the currently unread child Agent text (including the drop marker); clears the buffer. */ + drainText(): string { + return this.textBuffer.drain(); + } + + /** External wakeup (e.g. the parent tool call was aborted): makes a waiting `waitWake` return immediately. */ + wakeup(): void { + this.wakeSignal.notify(); + } + + /** Waits for "woken up" or `ms` to expire, whichever comes first. */ + async waitWake(ms: number): Promise { + await this.wakeSignal.wait(ms); + } + + /** + * Attaches an approval sink: the parent tool call active within the window hands in its own + * `ctx.approve`, and queued approval requests are consulted with Human through it one at a + * time. Returns a detach function (called when the window ends); a later attach replaces the + * former one. + */ + attachApprovalSink(approve: ApproveFn): () => void { + const epoch = ++this.sinkEpoch; + let onDetach!: () => void; + const detached = new Promise((resolve) => { + onDetach = resolve; + }); + this.sink = { approve, detached }; + void this.pumpApprovals(); + return () => { + if (this.sinkEpoch === epoch) this.sink = null; + onDetach(); + }; + } + + /** Cleanup: aborts the current run, denies pending approvals, releases child Session resources; idempotent. */ + kill(): void { + if (this.killed) return; + this.killed = true; + this.abortCtrl.abort(); + for (const req of [...this.approvals]) this.settle(req, "deny"); + // If running, released by pump's finally after it finishes; otherwise released immediately. + if (!this.isRunning) this.handle.dispose(); + this.wakeSignal.notify(); + } + + /** Synchronous hard-kill path: the child Session runs in-process with no separate OS resources, so this is equivalent to `kill`. */ + killHard(): void { + this.kill(); + } + + // ------------------------------------------------------------------------- + // Internal: pump and buffering + // ------------------------------------------------------------------------- + + /** Drives one round of `handle.run`: buffers messages and text, settling the terminal state when it ends. */ + private async pump(prompt: string): Promise { + let wroteAny = false; + let childAbort: string | null = null; + try { + for await (const msg of this.handle.run({ + prompt, + signal: this.abortCtrl.signal, + approve: this.childApprove, + })) { + this.bufferMessage(msg); + if ((msg.origin?.length ?? 0) === 1) { + const p = msg.payload as { + type?: string; + event_type?: string; + text?: string; + reason?: string; + }; + // A direct child layer's abort event: the child session was interrupted/failed (LLM + // request error, user interruption, etc). A child session failure doesn't throw, it + // only emits an event, based on which this round is reported as failed rather than + // marked completed. + if (p.type === "abort") { + childAbort = p.reason ?? "aborted"; + } else if ( + p.type === "partial_text" && + p.event_type === "delta" && + typeof p.text === "string" && + p.text + ) { + wroteAny = true; + this.appendText(p.text); + } + } + this.wakeSignal.notify(); + } + if (childAbort !== null) { + this.exitInfo = { status: "failed", note: `[subagent aborted: ${childAbort}]` }; + } else if (!wroteAny) { + this.exitInfo = { + status: "completed", + note: "[subagent finished without a text answer]", + }; + } else { + this.exitInfo = { status: "completed" }; + } + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + this.exitInfo = { status: "failed", note: `[subagent error: ${message}]` }; + } finally { + this.isRunning = false; + if (this.killed) this.handle.dispose(); + this.wakeSignal.notify(); + } + } + + private bufferMessage(msg: OmniMessage): void { + this.messages.push(msg); + // Overflow drops the oldest: only affects frontend replay — the child Session's Trace and text buffer are unaffected. + if (this.messages.length > MESSAGE_BUFFER_CAP) this.messages.shift(); + } + + private appendText(text: string): void { + this.textBuffer.append(text); + } + + // ------------------------------------------------------------------------- + // Internal: approval queue + // ------------------------------------------------------------------------- + + /** Approval callback handed to the child Session: the request is queued and waits for some parent tool call to consult Human and give a decision. */ + private readonly childApprove: ApproveFn = (toolCall) => { + if (this.killed) return Promise.resolve("deny"); + return new Promise((resolve) => { + this.approvals.push({ toolCall, settled: false, resolve }); + this.wakeSignal.notify(); // Wake the parent tool call waiting within the window, so it can consult as soon as possible + void this.pumpApprovals(); + }); + }; + + /** Settles an approval decision: first to arrive wins, late/duplicate decisions are ignored. */ + private settle(req: PendingApproval, decision: ApprovalDecision): void { + if (req.settled) return; + req.settled = true; + const idx = this.approvals.indexOf(req); + if (idx >= 0) this.approvals.splice(idx, 1); + req.resolve(decision); + this.wakeSignal.notify(); + } + + /** + * Hands the request at the head of the queue to the currently attached approval sink, one at a + * time. Stops when the window ends (the sink is detached); unresolved requests stay queued for + * the next sink; if a consultation already in flight resolves late, the decision still takes + * effect via settle. + */ + private async pumpApprovals(): Promise { + if (this.pumpingApprovals) return; + this.pumpingApprovals = true; + try { + while (this.sink && this.approvals.length > 0) { + const sink = this.sink; + const req = this.approvals[0]!; + const answer = sink.approve(req.toolCall).then( + (d) => this.settle(req, d), + () => this.settle(req, "deny"), // An approval sink error is treated as a denial (avoids leaving the child session stuck forever) + ); + await Promise.race([answer, sink.detached]); + if (req.settled) continue; + if (this.sink && this.sink !== sink) continue; // The sink was replaced by a new call: retry with the new sink + break; // The sink was detached and still unresolved: stay queued for the next sink + } + } finally { + this.pumpingApprovals = false; + } + } +} + +/** Converts a run's terminal state into a tool result (note is appended outside the truncation, so it isn't lost with long output). */ +export function resultForSubagentExit(exit: SubagentExit | null): ToolResult { + if (!exit) return { stopReason: "completed" }; + return { stopReason: exit.status, ...(exit.note !== undefined ? { note: exit.note } : {}) }; +} diff --git a/packages/core/src/environment/tools/types.ts b/packages/core/src/environment/tools/types.ts new file mode 100644 index 0000000..8a98755 --- /dev/null +++ b/packages/core/src/environment/tools/types.ts @@ -0,0 +1,79 @@ +/** + * BuiltinTool abstraction — lets Environment avoid special-casing any specific tool name. + * + * Each builtin tool carries its own: `name`, the `definition` handed to the LLM, and a streaming + * `execute`. Environment dispatches purely by looking up `name`; an unknown tool collapses to an + * explanatory `tool_call_output`, never throwing. Adding a new tool later (e.g. file read/write, + * retrieval) only requires implementing this interface and registering it with the registry, with + * no changes needed to Environment. + */ +import type { OmniMessage, StopReason } from "../../omnimessage/index.js"; +import type { ApproveFn, ToolDefinitionConfig } from "../../interfaces.js"; + +/** + * Tool execution context: runtime information needed to execute one tool call. + * Docs: /docs/tools § "Execution contract". + */ +export interface ToolExecutionContext { + /** Workspace absolute path; relative-path arguments should be resolved against it. */ + workspaceDir: string; + /** The tool_call_id passed through unchanged, used to build streaming deltas and nested origin tags. */ + toolCallId: string; + /** Abort signal; the tool should close out and return as soon as possible once it fires. */ + signal?: AbortSignal; + /** The parent Agent's approval callback; run_subagent passes it through to the child Session so it inherits the parent's approval mode (unused by most tools). */ + approve?: ApproveFn; +} + +/** + * Tool execution result (the generator's return value); treated as `completed` if omitted. + * Docs: /docs/tools § "Execution contract". + */ +export interface ToolResult { + stopReason?: StopReason; + /** + * Terminal marker (e.g. `[exit code: 1]`): appended by Environment during its unified + * close-out, **outside** the maxOutputLength truncation, and streamed to the frontend as an + * extra chunk — so the failure marker isn't lost when long output gets truncated (it would be + * cut off if produced as a content delta instead). + */ + note?: string; + /** + * Images carried by the tool output (e.g. an image read by read_image): each entry is a + * `data:;base64,...` data URL. Attached by Environment during close-out: a single + * streaming delta carries it all at once before stop, plus the final complete + * `tool_call_output` (only carried on normal completion; images are not chunked and don't + * count toward text truncation). + */ + images?: string[]; +} + +/** + * Builtin tool interface. `execute` receives the already-parsed tool argument object and the + * execution context, streaming out OmniMessage as an async generator. Contract (a relaxed + * version — framing and close-out are handled uniformly by Environment): + * + * - **Own output**: yielding the **delta** of `partial_tool_call_output` is enough; `start`/`stop` + * are optional (Environment ignores the tool's start/stop and frames it itself), and there's + * **no need** to produce a complete `tool_call_output` either (the complete message, + * maxOutputLength forward truncation, and close-out are all derived by Environment from the + * deltas). If a tool does produce a complete `tool_call_output` anyway, Environment uses it as + * the basis for content and stop reason (tolerated for compatibility, not recommended). + * - **Nested forwarding**: yielding any message **tagged with origin** is passed through by + * Environment unchanged (e.g. run_subagent forwarding all of a child session's messages). + * - **Stop reason**: reported via the generator's return value (defaults to completed); a throw + * is collapsed by Environment into aborted/failed based on interruption/error, never + * propagating up as an exception. + * Docs: /docs/interfaces § "The inner tool contract: BuiltinTool"; /docs/tools § "Execution contract". + */ +export interface BuiltinTool { + /** Tool name (corresponds to the tool_call.name returned by the LLM). */ + name: string; + /** Tool definition handed to the LLM (including description / parameters / permission / maxOutputLength). */ + definition: ToolDefinitionConfig; + /** Executes one tool call: args is the already-parsed argument object, ctx is the runtime context. */ + execute( + args: Record, + ctx: ToolExecutionContext, + ): AsyncGenerator; +} diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts new file mode 100644 index 0000000..28f40d4 --- /dev/null +++ b/packages/core/src/index.ts @@ -0,0 +1,45 @@ +/** + * @prismshadow/penguin-core — public entry point for the PenguinHarness core SDK. + * + * Exports the OmniMessage protocol, the three interface contracts (Human/LLM/Environment), + * and the runtime entry points for Agent / Session / context_engine along with their + * submodules (state / llm / environment / trace). + * + * Typical usage: + * + * ```ts + * const agent = await createAgent({ agentId: "default_agent" }); + * const session = await agent.createSession({ workspaceDir, modelId }); + * for await (const output of session.run([userText("...")])) { ... } + * ``` + */ + +// Protocol and interface contracts (foundation) +export * from "./omnimessage/index.js"; +export * from "./interfaces.js"; + +// Submodules +export * from "./state/index.js"; +export * from "./llm/index.js"; +export * from "./environment/index.js"; +export * from "./trace/index.js"; + +// Runtime entry points +export { ContextEngine } from "./engine/context-engine.js"; +export type { + CompactAvailability, + CompactionSettings, + ContextEngineDeps, + EngineInitialState, + RunOptions, + TraceSink, +} from "./engine/context-engine.js"; +export { Session } from "./session.js"; +export type { SessionConfig } from "./session.js"; +export { buildTitlePrompt, generateTitleWithLLM, sanitizeTitle } from "./session-title.js"; +export type { SessionTitleResult } from "./session-title.js"; +export { Agent, createAgent } from "./agent.js"; +export type { CreateAgentOptions, CreateSessionOptions, ResumeSessionOptions } from "./agent.js"; + +/** SDK version number. */ +export const VERSION = "0.0.1"; diff --git a/packages/core/src/interfaces.ts b/packages/core/src/interfaces.ts new file mode 100644 index 0000000..c8c463c --- /dev/null +++ b/packages/core/src/interfaces.ts @@ -0,0 +1,279 @@ +/** + * Internal SDK interface contracts: LLM, Environment. + * + * `context_engine` only handles OmniMessage; protocol conversion and concrete implementations + * are each interface's own responsibility. + * Human is not an "interface/class with methods" but the SDK's input/output boundary itself: + * output is streamed by `Session.run()` as an async generator, and input is delivered via + * `run`'s `RunOptions` — approvals are requested one at a time through the injected `approve` + * callback, and interruption goes through `signal`. Hence no Human interface is defined here. + * + * These types form the foundational contract shared by all units; implementing units integrate + * against them. + * + * Docs: packages/docs/content/interfaces.{zh,en}.md (site path /docs/interfaces) explains each + * contract and its extension seams — keep the page in sync when changing signatures here. + */ +import type { + ApprovalDecision, + OmniMessage, + StopReason, + ToolCallPayload, + ToolDefinition, +} from "./omnimessage/types.js"; +// Concrete classes, used only for EnvironmentServices type annotations (type-only import; no runtime dependency, no circular reference). +import type { CommandSessionManager } from "./environment/tools/command/session-manager.js"; +import type { SubagentSessionManager } from "./environment/tools/subagent/session-manager.js"; +import type { ToolCallIdAllocator } from "./llm/tool-call-ids.js"; + +// --------------------------------------------------------------------------- +// Tool definitions and configuration +// --------------------------------------------------------------------------- + +// ToolDefinition is defined in omnimessage/types.ts (session_meta embeds the full tool schema directly); re-exported here to keep the original import path. +export type { ToolDefinition } from "./omnimessage/types.js"; + +/** Tool permission: read-only / read-write. */ +export type ToolPermission = "r" | "rw"; + +/** + * Runtime configuration for a single tool. + * Docs: /docs/tools § "Configuration fields". + */ +export interface ToolDefinitionConfig { + name: string; + description: string; + parameters?: Record; + permission?: ToolPermission; + /** + * Which class of session model this entry targets: `"vision"` only for models that support + * images (e.g. read_image), `"text-only"` only for text-only models (e.g. describe_image); + * omitted means available for all models. Filtered by session model at assembly time + * (see `selectBuiltinToolsForModel`). + */ + forModel?: "vision" | "text-only"; + /** Timeout for a single tool call (ms); on timeout, ends as `failed`; <=0 disables it. */ + timeoutMs?: number; + /** Max length of tool output; Environment truncates from the front (keeping the head) if exceeded; <=0 disables it. */ + maxOutputLength?: number; +} + +export interface MCPServerConfig { + name: string; + config: Record; +} + +/** Set of tool configs required to initialize Environment. */ +export interface ToolConfig { + customTools: ToolDefinitionConfig[]; + mcpServers: MCPServerConfig[]; +} + +/** + * Per-tool approval callback: the Human boundary gives allow/deny for each complete `tool_call`. + * `context_engine` calls it once per tool call within a turn. Subagents forward the parent's + * approval callback, so the child Agent **inherits the parent Agent's approval mode**. + * Docs: /docs/interfaces § "ApproveFn". + */ +export type ApproveFn = (toolCall: OmniMessage) => Promise; + +// --------------------------------------------------------------------------- +// LLM interface +// --------------------------------------------------------------------------- + +export type ThinkingLevelName = "none" | "low" | "medium" | "high" | "xhigh"; + +/** + * GenerativeModel initialization config. + * Docs: /docs/interfaces § "GenerativeModelConfig". + */ +export interface GenerativeModelConfig { + modelId: string; + apiKey?: string; + baseUrl?: string; + /** + * AgentHub client protocol (`openai` / `claude-4-8` / `deepseek-v4` / …). If omitted, AgentHub + * infers it from `modelId`; custom-named models or third-party models using the OpenAI protocol + * must specify it explicitly. + */ + clientType?: string; + tools: ToolDefinition[]; + /** Full system Prompt after placeholder substitution in the system_config.system_prompt template. */ + systemPrompt?: string; + contextWindow?: number; + maxTokens?: number; + thinkingLevel?: ThinkingLevelName; + /** LLM Request timeout (ms): from system_config.model.timeoutMs; <=0 disables it. Defaults to 120000. */ + requestTimeoutMs?: number; + /** + * tool_call_id uniqueness registry (Session-level). Pass the same instance when rebuilding a new + * GenerativeModel on compaction so the uniqueness scope covers the whole Session; defaults to a fresh + * one. See llm/tool-call-ids.ts. + */ + toolCallIds?: ToolCallIdAllocator; +} + +export interface GenerativeModelParameters { + /** OmniMessage array for the input newly added this turn; implementations must merge it into a single UniMessage (multiple roles not accepted). */ + newMessages: OmniMessage[]; + signal?: AbortSignal; +} + +/** + * The terminal state of an LLM request, returned as the **return value** of the `streamGenerate` + * async generator (not a yielded message). The status values share the same five-value protocol + * as OmniMessage `stop_reason`: + * - `completed`: finished normally (already produced `token_usage`); + * - `timeout`: LLM timed out or lost connection, needs reconnect — retried by `context_engine` + * within the same run; + * - `malformed`: AgentHub response failed JSON parsing, needs reconnect — also retried by + * `context_engine`; + * - `aborted`: user-initiated interruption — stop and hand back to the user; + * - `failed`: other non-retryable errors (auth/params, etc.) — stop and hand back to the user + * (`message` provides the display text). + * Docs: /docs/interfaces § "LLMOutcome semantics". + */ +export interface LLMOutcome { + status: StopReason; + message?: string; +} + +/** + * A stateful LLM object attached to a Session. + * `streamGenerate` yields streaming `partial_*` messages as an async generator, and appends the + * corresponding complete `model_msg` once each fragment ends; Token usage is emitted as a + * `token_usage` event_msg. **Never throws to `context_engine`**: any interruption/exception is + * closed off in well-formed structure and returned normally, and **must** report the terminal + * state via `LLMOutcome` — error handling happens entirely inside the LLM interface, and + * `context_engine` only decides subsequent actions based on the outcome. + * Docs: /docs/interfaces § "LLMInterface". + */ +export interface LLMInterface { + streamGenerate(parameters: GenerativeModelParameters): AsyncGenerator; +} + +// --------------------------------------------------------------------------- +// Environment interface +// --------------------------------------------------------------------------- + +/** + * Handle for a child Agent session: derived by `SubagentRunner.spawn`, + * representing a child Session that can run over multiple turns. Deriving (spawn) is separate + * from running (run), so the same child Session can accept an additional Prompt and keep running + * after a turn ends (a long-running subagent, accessed via `input_subagent`). + * Docs: /docs/interfaces § "Subagent interfaces". + */ +export interface SubagentHandle { + /** The child Session's id: the origin hop of messages produced by run; `subagent_id` is derived from its tail for the frontend to correlate. */ + sessionId: string; + /** + * Runs one turn of a task on the child Session. Emitted child-session messages **all already + * carry the origin marker** (the child Session id); the first message of the first run is the + * child Session's `session_meta`, and tool_calls received by the forwarded approval callback + * carry origin as well. + */ + run(input: { + /** The task Prompt handed to the child Agent. */ + prompt: string; + signal?: AbortSignal; + /** The parent Agent's approval callback; forwarded to the child Session to inherit the parent's approval mode. */ + approve?: ApproveFn; + }): AsyncGenerator; + /** Releases runtime resources held by the child Session (e.g. its managed command sessions). Idempotent. */ + dispose(): void; +} + +/** + * Child Agent runner: injected into the `run_subagent` tool so it can + * derive and run a child Agent without a reverse dependency on Agent/Session, avoiding circular + * dependencies. The concrete implementation is provided by the SDK composition layer (where + * `createAgent` lives), which internally derives via `createAgent` → `createSession` and hands + * back a `SubagentHandle`. + * Docs: /docs/interfaces § "Subagent interfaces". + */ +export interface SubagentRunner { + /** + * Derives a child Agent and creates a child Session. Precheck errors such as exceeding the + * depth limit or a nonexistent target agent are expressed by throwing (collapsed to `failed` + * by Environment). + */ + spawn(input: { + /** The child Agent's agentId; if omitted, reuses the current Agent (self-invocation). */ + agentId?: string; + /** The Model used by the child Session; if omitted, uses the Project's default Model. */ + modelId?: string; + }): Promise; +} + +/** + * Proxy-reading service for describe_image: injected when the session model doesn't support + * images (vision=false) — images are handed to the configured vision model for description and + * the tool returns text, avoiding a 400 from feeding images back into a tool_result for a + * provider that doesn't support images. + * Docs: /docs/interfaces § "VisionDescriberService". + */ +export interface VisionDescriberService { + /** Vision model id; null when the Project has no `vision_model` configured (or it's invalid), in which case the tool ends with a failed explanation. */ + modelId: string | null; + /** Constructs a single-shot LLM for this vision model (no tools, no system prompt); omitted when `modelId` is null. */ + createLLM?: () => LLMInterface; +} + +/** + * Runtime services Environment injects into individual tools (e.g. `run_subagent` needs `SubagentRunner`); most tools don't use these. + * Docs: /docs/interfaces § "ToolExecutionRequest and EnvironmentConfig". + */ +export interface EnvironmentServices { + subagentRunner?: SubagentRunner; + /** Injected when the session model doesn't support images: for describe_image's single-shot vision-model proxy reading. */ + visionDescriber?: VisionDescriberService; + /** Registry of long-running command sessions (shared by `exec_command` / `input_command`); constructed and injected internally by Environment. */ + commandSessions?: CommandSessionManager; + /** Registry of background subagent sessions (shared by `run_subagent` / `input_subagent`); constructed and injected internally by Environment. */ + subagentSessions?: SubagentSessionManager; +} + +/** Docs: /docs/interfaces § "ToolExecutionRequest and EnvironmentConfig". */ +export interface EnvironmentConfig { + workspaceDir: string; + toolConfig: ToolConfig; + /** Runtime services (optional); Environment forwards these to each tool factory to use as needed. */ + services?: EnvironmentServices; + /** + * Agent vault environment variables (key-value pairs, taken from the Agent's + * `agent_state/.vault.toml`): injected into the exec_command / input_command subprocess + * environment; hardened entries cannot be overridden. + */ + vault?: Record; +} + +/** + * An approved tool-call execution request. + * Docs: /docs/interfaces § "ToolExecutionRequest and EnvironmentConfig". + */ +export interface ToolExecutionRequest { + /** The OmniMessage whose payload.type === "tool_call". */ + toolCall: OmniMessage; + signal?: AbortSignal; + /** The parent Agent's approval callback; forwarded to tools that need to derive a child Session (run_subagent), implementing approval inheritance. */ + approve?: ApproveFn; +} + +/** + * Environment interface: executes approved tool calls within the Workspace. + * `executeTool` yields `partial_tool_call_output` as an async generator and ends with exactly one + * complete `tool_call_output`; nested session messages carrying an origin marker (e.g. forwarded + * by run_subagent) pass through unchanged. + * + * **Rendering** of tool calls is not this interface's concern (nor core's): streaming rendering is + * handled by the CLI / Web frontend itself. + * Docs: /docs/interfaces § "EnvironmentInterface". + */ +export interface EnvironmentInterface { + listTools(): Promise; + executeTool(request: ToolExecutionRequest): AsyncGenerator; + /** Looks up a tool's permission level (for frontend permission-mode decisions); returns undefined for unknown tools. */ + toolPermission(name: string): ToolPermission | undefined; + /** Releases runtime resources held by the environment (e.g. managed long-running command sessions); called by the host when the Session ends. Optional, idempotent. */ + dispose?(): void; +} diff --git a/packages/core/src/internal/dates.ts b/packages/core/src/internal/dates.ts new file mode 100644 index 0000000..eef77b0 --- /dev/null +++ b/packages/core/src/internal/dates.ts @@ -0,0 +1,9 @@ +/** Local-timezone date formatting (internal shared helper, not exported via the barrel). */ + +/** Format a date as local `yyyy-mm-dd` (local timezone, 4-digit year, zero-padded 2-digit month/day). */ +export function formatLocalDate(date: Date): string { + const year = date.getFullYear().toString().padStart(4, "0"); + const month = (date.getMonth() + 1).toString().padStart(2, "0"); + const day = date.getDate().toString().padStart(2, "0"); + return `${year}-${month}-${day}`; +} diff --git a/packages/core/src/internal/session-support.ts b/packages/core/src/internal/session-support.ts new file mode 100644 index 0000000..9e6484e --- /dev/null +++ b/packages/core/src/internal/session-support.ts @@ -0,0 +1,171 @@ +/** + * Session creation helpers (used by `agent.createSession` for assembly, not exported + * via the barrel): Session id generation, runtime environment fields, and temp + * Workspace creation. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { randomBytes, randomUUID } from "node:crypto"; + +import { formatLocalDate } from "./dates.js"; +import type { SessionEnvironmentValues } from "../state/agent-state.js"; +import { workspacesDir } from "../state/index.js"; +import { userText } from "../omnimessage/index.js"; +import type { OmniMessage } from "../omnimessage/index.js"; + +/** Session runtime environment fields: the placeholder substitution values for `assembleSystemPrompt`; producer and consumer share the same type. */ +export type SessionEnvironment = SessionEnvironmentValues; + +/** Generate a Session id of the form `session-YYYY-MM-DD-HH-mm-ss-<8-hex>` (local timezone, zero-padded: 4-digit year, 2 digits for the rest; hex from randomUUID). */ +export function formatSessionId(date: Date = new Date()): string { + const pad = (n: number) => n.toString().padStart(2, "0"); + const ts = + `${formatLocalDate(date)}` + + `-${pad(date.getHours())}-${pad(date.getMinutes())}-${pad(date.getSeconds())}`; + const hex = randomUUID().replace(/-/g, "").slice(0, 8); + return `session-${ts}-${hex}`; +} + +/** + * Generate this Session's runtime environment fields (injected via specific + * placeholders in the system prompt). + * This is system-generated runtime context, not sourced from Agent State / Workspace files. + */ +export function sessionEnvironment( + workspaceDir: string, + sessionId: string, + ids: { agentId: string; projectDir: string }, + date = new Date(), +): SessionEnvironment { + return { + sessionId, + cwd: workspaceDir, + agentId: ids.agentId, + projectDir: ids.projectDir, + platform: process.platform, + osVersion: getOsVersion(), + date: formatLocalDate(date), + }; +} + +function getOsVersion(): string { + // os.* is a stable built-in API that normally doesn't throw; but this function only + // builds a single line of environment info for the system prompt, so it's not worth + // letting an exception take down createSession — fall back to "unknown" instead. + try { + if (process.platform === "win32") { + return `${os.version()} ${os.release()}`; + } + return `${os.type()} ${os.release()}`; + } catch { + return "unknown"; + } +} + +/** The 8-hex space is 2^32, so the odds of consecutive collisions are negligible; the cap only guards against an infinite loop caused by an abnormal filesystem. */ +const MAX_TMP_ID_ATTEMPTS = 16; + +/** + * Create a temporary Workspace under `/workspaces/`, where the + * directory name is the workspace_id, shaped like `tmp-<8hex>`; if it collides with + * an existing directory, regenerate the id. No symlinks are created inside the Workspace: + * the model composes absolute paths (to Agent State, scratchpad, etc.) directly from the + * Environment placeholders (Project Dir / Agent ID) in the system prompt. + */ +export async function createTempWorkspace( + root: string, + projectId: string, + agentId: string, +): Promise { + const base = workspacesDir(root, projectId, agentId); + await fs.mkdir(base, { recursive: true }); + // The final directory must use a non-recursive mkdir: recursive mkdir succeeds + // silently when the directory already exists, which would put a new Session into + // an existing temp Workspace; EEXIST means an id collision, so retry with a new id. + for (let attempt = 0; attempt < MAX_TMP_ID_ATTEMPTS; attempt++) { + const dir = path.join(base, `tmp-${randomUUID().slice(0, 8)}`); + try { + await fs.mkdir(dir); + return dir; + } catch (err) { + if ((err as NodeJS.ErrnoException).code !== "EEXIST") throw err; + } + } + throw new Error( + `failed to allocate a unique temp workspace id under ${base} after ${MAX_TMP_ID_ATTEMPTS} attempts`, + ); +} + +/** Maps a data URL's mime type to a file extension on disk; unknown mimes use bin (the image-reading tool sniffs the magic bytes and doesn't rely on the extension). */ +const MIME_TO_EXT: Record = { + "image/png": "png", + "image/jpeg": "jpg", + "image/gif": "gif", + "image/webp": "webp", +}; + +/** + * Input conversion for when the session model doesn't support images: image messages + * in the `run` input are written to disk as files (base64 data URLs are saved to the + * session scratchpad; http(s) URLs are referenced + * as-is), and the path/URL is appended to the user text (an `[attached image: …]` + * line); the image message itself is removed from the input — the model views it by + * path via describe_image (read on its behalf by a vision model), and images never + * enter that session's history directly. + * Returns the input unchanged when there are no images; an image that can't be + * parsed is replaced with an explanatory line rather than silently dropped. + */ +export async function imagesToScratchpadPaths( + input: OmniMessage[], + dir: string, +): Promise { + const isImage = (m: OmniMessage): boolean => + (m.payload as { type?: string }).type === "image_url"; + if (!input.some(isImage)) return input; + + const lines: string[] = []; + for (const msg of input) { + if (!isImage(msg)) continue; + const url = (msg.payload as { image_url?: string }).image_url ?? ""; + if (/^https?:\/\//i.test(url)) { + lines.push(`[attached image: ${url}]`); + continue; + } + const match = /^data:([^;,]+);base64,(.+)$/s.exec(url); + if (!match) { + lines.push("[an attached image could not be saved and was dropped]"); + continue; + } + await fs.mkdir(dir, { recursive: true }); + const ext = MIME_TO_EXT[match[1]!.toLowerCase()] ?? "bin"; + // Filename = upload-<8 random hex chars> (same convention as project-<8hex>; the + // prefix distinguishes model-generated temp files). + // "wx" flag does exclusive creation to avoid name collisions: on the rare chance of a collision, retry with a new random value. + let file: string; + for (;;) { + file = path.join(dir, `upload-${randomBytes(4).toString("hex")}.${ext}`); + try { + await fs.writeFile(file, Buffer.from(match[2]!, "base64"), { flag: "wx" }); + break; + } catch (err) { + if ((err as NodeJS.ErrnoException).code !== "EEXIST") throw err; + } + } + lines.push(`[attached image: ${file}]`); + } + + // Concatenation: the path lines are appended after the last user text message; if the input is images only, add a plain path-only text message. + const rest = input.filter((m) => !isImage(m)); + const suffix = lines.join("\n"); + const lastTextIdx = rest.findLastIndex((m) => { + const p = m.payload as { type?: string; role?: string }; + return p.type === "text" && p.role === "user"; + }); + if (lastTextIdx === -1) return [...rest, userText(suffix)]; + return rest.map((m, i) => { + if (i !== lastTextIdx) return m; + const p = m.payload as { type: string; role: string; text: string }; + return { ...m, payload: { ...p, text: `${p.text}\n\n${suffix}` } } as OmniMessage; + }); +} diff --git a/packages/core/src/llm/generative-model.ts b/packages/core/src/llm/generative-model.ts new file mode 100644 index 0000000..87f7a0c --- /dev/null +++ b/packages/core/src/llm/generative-model.ts @@ -0,0 +1,1059 @@ +/** + * GenerativeModel —— the SDK's LLM interface implementation. + * + * Responsibilities (protocol translation + streaming aggregation): + * 1. Merge a group of OmniMessages that **share the same role** into a single AgentHub `UniMessage`; + * 2. Issue the request via `AutoLLMClient.streamingResponseStateful` (stateful — AgentHub + * maintains history internally), translating streamed `UniEvent`s back into OmniMessages: + * - text/thinking/tool-call deltas → `partial_*` (before the first delta of each segment + * `yield` a `start`; after the segment ends `yield` a `stop`); + * - after a segment ends, append the full `model_msg` (thinking / text / tool_call); + * - a `token_usage` event_msg is produced **only on normal completion** + * (observability/Token). + * 3. Interruption/error handling: `finishInterrupted` first closes any open + * streaming segments and backfills the complete message, then the output ends — never + * leaking a malformed structure. This interface **never retries internally** — retryable + * errors (network/timeout/429/5xx, see `isRetryableError`) end with `timeout`; AgentHub + * JSON parse errors end with `malformed`; both are handed to `context_engine` to reconnect + * within the same run. User interruption ends with `aborted`; non-retryable errors + * (auth/parameters) end with `failed`. + * + * `context_engine` only consumes OmniMessage; all Uni* protocol details are encapsulated here. + * Docs: /docs/interfaces § "The built-in implementation: GenerativeModel". + */ +import { AutoLLMClient, ThinkingLevel } from "@prismshadow/agenthub"; +import type { + ContentItem, + FinishReason, + ToolSchema, + UniConfig, + UniEvent, + UniMessage, + UsageMetadata, +} from "@prismshadow/agenthub"; + +import { + addTokenCounts, + assistantText, + emptyTokenCounts, + partialText, + partialThinking, + partialToolCall, + thinkingMessage, + tokenUsage, + toolCall, +} from "../omnimessage/index.js"; +import type { + CompleteModelPayload, + OmniMessage, + StopReason, + TokenCounts, +} from "../omnimessage/index.js"; +import type { + GenerativeModelConfig, + GenerativeModelParameters, + LLMInterface, + LLMOutcome, + ThinkingLevelName, + ToolDefinition, +} from "../interfaces.js"; +import { ToolCallIdAllocator, stripToolCallIdSuffix } from "./tool-call-ids.js"; + +// --------------------------------------------------------------------------- +// Pure conversion function: OmniMessage[] → a single UniMessage (unit-testable, no network) +// --------------------------------------------------------------------------- + +/** + * Tool arguments JSON string → object. History only ever contains tool_calls from committed + * turns (non-completed turns are discarded on replay); bad JSON already throws during AgentHub + * parsing and reconnects via malformed, so it never enters history — hence we parse directly + * with no fallback tolerance; an empty string is treated as no arguments. + */ +function parseToolArguments(raw: string): Record { + if (!raw) return {}; + const parsed: unknown = JSON.parse(raw); + return parsed !== null && typeof parsed === "object" ? (parsed as Record) : {}; +} + +/** + * Maps a complete OmniMessage payload to an AgentHub `ContentItem`. + * Only complete model_msg payloads are supported; `partial_*` is an output-only protocol. + */ +function payloadToContentItem(payload: CompleteModelPayload): ContentItem { + // Provider-fidelity fields (signature / phase) are restored verbatim — some models require + // them when history is replayed back (e.g. Claude thinking signatures, GPT-5 encrypted + // reasoning and phase segmentation); losing them would break Session recovery. + switch (payload.type) { + case "text": + return { + type: "text", + text: payload.text, + ...(payload.phase != null ? { phase: payload.phase } : {}), + ...(payload.signature !== undefined ? { signature: payload.signature } : {}), + }; + case "image_url": + return { type: "image_url", image_url: payload.image_url }; + case "inline_data": + return { + type: "inline_data", + data: Buffer.from(payload.data, "base64"), + mime_type: payload.mime_type, + ...(payload.signature !== undefined ? { signature: payload.signature } : {}), + }; + case "inline_thinking": + return { + type: "inline_thinking", + data: Buffer.from(payload.data, "base64"), + mime_type: payload.mime_type, + ...(payload.signature !== undefined ? { signature: payload.signature } : {}), + }; + case "thinking": + return { + type: "thinking", + thinking: payload.thinking, + ...(payload.signature !== undefined ? { signature: payload.signature } : {}), + }; + case "tool_call": + return { + type: "tool_call", + name: payload.name, + // OmniMessage stores arguments as a JSON string; UniMessage uses an object. + arguments: parseToolArguments(payload.arguments), + // On the way back, strip the uniqueness suffix to restore the provider's original id (see tool-call-ids.ts). + tool_call_id: stripToolCallIdSuffix(payload.tool_call_id), + ...(payload.signature !== undefined ? { signature: payload.signature } : {}), + }; + case "tool_call_output": + return { + type: "tool_result", + text: payload.output, + // Images carried by the tool output (data URL array) → AgentHub tool_result.images + // (natively supported). + ...(payload.images && payload.images.length > 0 ? { images: payload.images } : {}), + // Gemini pairs by using tool_call_id as functionResponse.name, so it must be restored to the original id (the function name). + tool_call_id: stripToolCallIdSuffix(payload.tool_call_id), + }; + default: { + // Exhaustiveness check: compile-time error when a new payload type is added. + const _exhaustive: never = payload; + throw new Error( + `streamGenerate: unsupported message type: ${(_exhaustive as { type?: string }).type}`, + ); + } + } +} + +/** + * Groups a replayed, complete OmniMessage history into a sequence of UniMessages by + * **adjacent same-role** runs (used for the setHistory injection during Session recovery). + * One committed turn = a group of user-side input + a group of + * assistant output, matching exactly the adjacent user / assistant UniMessages in AgentHub history. + */ +export function groupHistoryToUniMessages(history: OmniMessage[]): UniMessage[] { + const groups: OmniMessage[][] = []; + let currentRole: string | null = null; + for (const msg of history) { + const role = (msg.payload as { role?: string }).role; + if (role !== "user" && role !== "assistant") { + throw new Error(`setHistory: unsupported message without role: ${JSON.stringify(msg.type)}`); + } + if (role !== currentRole) { + groups.push([]); + currentRole = role; + } + groups[groups.length - 1]!.push(msg); + } + return groups.map(mergeOmniToUniMessage); +} + +/** + * Merges a group of OmniMessages into **a single** UniMessage. + * + * Constraint: all messages in the array must share the same + * role; the role of the first payload is used as the UniMessage's role (a tool_call_output + * group has role "user"). Throws if roles are mixed. + */ +export function mergeOmniToUniMessage(messages: OmniMessage[]): UniMessage { + if (messages.length === 0) { + throw new Error("streamGenerate requires at least one input message"); + } + + const payloads = messages.map((m) => m.payload as CompleteModelPayload); + // Each payload carries its own role (tool_call_output is fixed to "user"); take the first one's role. + const role = payloads[0]!.role; + + const contentItems: ContentItem[] = []; + for (const payload of payloads) { + if (payload.role !== role) { + throw new Error( + "streamGenerate does not accept mixed roles: all messages merged into one UniMessage must share the same role", + ); + } + contentItems.push(payloadToContentItem(payload)); + } + + return { role, content_items: contentItems }; +} + +// --------------------------------------------------------------------------- +// Token accounting (as defined by SKILL.md) +// --------------------------------------------------------------------------- + +/** + * Converts AgentHub `UsageMetadata` into PenguinHarness `TokenCounts`. + * + * Conversion rules (AgentHub UsageMetadata → OmniMessage TokenCounts, null treated as 0): + * - `cache_read = cached_tokens` (input tokens served from cache hits); + * - `cache_write = prompt_tokens` (input tokens on a cache miss); + * - `output = thoughts_tokens + response_tokens`; + * - `total = cache_read + cache_write + output`. + * That is, `input = cache_read + cache_write = cached_tokens + prompt_tokens`, + * and `total = input + output` (consistent with SKILL.md's input/output accounting). + */ +export function usageToTokenCounts(usage: UsageMetadata): TokenCounts { + const cached = usage.cached_tokens ?? 0; + const prompt = usage.prompt_tokens ?? 0; + const thoughts = usage.thoughts_tokens ?? 0; + const response = usage.response_tokens ?? 0; + const cacheRead = cached; + const cacheWrite = prompt; + const output = thoughts + response; + return { + cache_read: cacheRead, + cache_write: cacheWrite, + output, + total: cacheRead + cacheWrite + output, + }; +} + +// --------------------------------------------------------------------------- +// Pure translator: UniEvent[] → OmniMessage[] (unit-testable, no network) +// --------------------------------------------------------------------------- + +interface ToolCallAccumulator { + name: string; + /** Accumulated arguments delta fragments (JSON string); used as a fallback when no complete tool_call arrives. */ + argsBuffer: string; + /** If a complete tool_call appears in an event, its JSON.stringify'd arguments are recorded here. */ + completeArgs: string | null; + /** Session-unique id emitted to OmniMessage (with a `#n` suffix on provider id collisions). */ + toolCallId: string; + /** The original tool_call_id reported by the provider (the attribution key for inbound events). */ + providerKey: string; + /** Provider-fidelity field: signature (kept verbatim, produced alongside the complete tool_call). */ + signature: string | undefined; + /** Whether this tool_call's complete message has already been emitted eagerly in `pushEvent` (avoids duplicate emission in finish). */ + emitted: boolean; +} + +/** + * Streaming translator. Feed `UniEvent`s one at a time into `pushEvent`, which yields + * incremental OmniMessages (`partial_*`); after the stream ends, call `finish` to produce + * `stop`, the complete `model_msg`, and `token_usage`. + * + * Split into its own class to make unit testing easier (feed a constructed array of UniEvents, + * assert on emission order / aggregation / token counts). + * Docs: /docs/omni-message § "The streaming discipline". + */ +export class EventTranslator { + /** + * tool_call_id uniqueness registry. By default each translator creates its own (unit tests / + * one-off translation); in production `GenerativeModel` injects a Session-level shared instance so + * the uniqueness scope spans Requests and survives compaction rebuilds. + */ + constructor(private readonly toolCallIds: ToolCallIdAllocator = new ToolCallIdAllocator()) {} + + // Whether each segment type has already yielded its `start`. + private textStarted = false; + private thinkingStarted = false; + // Tool-call partial starts, tracked by the (uniqueness-resolved) tool_call_id. + private toolStarted = new Set(); + // Some providers' tool argument deltas don't carry a tool_call_id; attribute them to the most recently opened tool call (by provider id key). + private activeToolCallId: string | null = null; + + // Buffers needed for the complete message. + private textBuffer = ""; + private thinkingBuffer = ""; + // Provider-fidelity fields: the thinking block's signature (a signature + // marks the end of a block) and the current text segment's phase (sticky across segments; + // a differing phase marker starts a new segment) and signature. + private thinkingSignature: string | undefined; + private textPhase: string | null = null; + private textSignature: string | undefined; + /** Provider id keys saved in order of appearance, so complete tool_calls are emitted in a stable order. */ + private toolOrder: string[] = []; + /** provider's original tool_call_id → the accumulator for the **latest** call under that id. */ + private tools = new Map(); + + private finishReason: FinishReason | null = null; + /** Token usage for this request (a snapshot from the most recent usage report). */ + private requestTokens: TokenCounts = emptyTokenCounts(); + + /** Consumes one UniEvent, yielding 0..n streaming OmniMessages. */ + *pushEvent(event: UniEvent): Generator { + if (event.finish_reason != null) { + this.finishReason = event.finish_reason; + } + if (event.usage_metadata) { + // The same request may report usage multiple times, always as a **cumulative snapshot** + // (Gemini reports one per chunk, as do some OpenAI-compatible endpoints; Claude/GPT-5, + // aggregated by AgentHub, report only once at the end). Overwrite with the latest snapshot + // — never accumulate: summing snapshots chunk by chunk would inflate usage by roughly the + // number of chunks. + this.requestTokens = this.usageOnce(event.usage_metadata); + } + + for (const item of event.content_items) { + switch (item.type) { + case "text": { + // Provider fidelity: a phase marker can arrive as an increment with **empty text** + // (e.g. GPT-5's segment markers). A phase differing from the current segment's phase + // starts a new segment — providers split by phase when replaying history, so mixing + // segments would break fidelity. Phase is sticky across segments (a subsequent segment + // with the same phase isn't re-marked). + if (item.phase != null && item.phase !== this.textPhase) { + yield* this.flushThinking("completed"); + yield* this.flushText("completed"); + this.textPhase = item.phase; + } + if (item.signature) this.textSignature = item.signature; + if (!item.text) break; + // Type boundary: before a text segment starts, flush any unclosed thinking + // segment, so the complete-message order matches generation order (thinking → text). + // The boundary flush uses completed: that thinking segment's stop reason is "switched + // to text", not "Request ended". + yield* this.flushThinking("completed"); + if (!this.textStarted) { + this.textStarted = true; + yield partialText("start"); + } + this.textBuffer += item.text; + yield partialText("delta", item.text); + break; + } + case "thinking": { + // Provider fidelity: a thinking block ends with a signature (Claude's signature_delta + // is empty text + signature; redacted blocks carry sentinel text + signature; GPT-5 + // encrypted reasoning is empty text + signature). If thinking content/a new signature + // arrives after a signature is already set, that's a new block — close the current + // segment first, so each block's signature stays independently faithful and blocks + // don't bleed into each other when history is replayed. + if (this.thinkingSignature !== undefined && (item.thinking || item.signature)) { + yield* this.flushThinking("completed"); + } + if (item.signature) this.thinkingSignature = item.signature; + if (!item.thinking) break; + // Type boundary: before a thinking segment starts, flush any unclosed text + // segment, so the complete-message order matches generation order (text → thinking). + // The boundary flush uses completed: that text segment's stop reason is "switched to + // thinking", not "Request ended". + yield* this.flushText("completed"); + if (!this.thinkingStarted) { + this.thinkingStarted = true; + yield partialThinking("start"); + } + this.thinkingBuffer += item.thinking; + yield partialThinking("delta", item.thinking); + break; + } + case "partial_tool_call": { + // Some providers' argument deltas don't carry a tool_call_id; attribute them to the most + // recently opened tool call. If there's no open tool call yet, skip — don't fabricate a + // tool_call with an empty id. + const providerKey = item.tool_call_id || this.activeToolCallId; + if (!providerKey) break; + if (item.tool_call_id) this.activeToolCallId = item.tool_call_id; + const acc = this.ensureTool(providerKey, item.name); + if (item.name) acc.name = item.name; + if (item.signature) acc.signature = item.signature; + // Externally always use the uniqueness-resolved id (with a `#n` suffix on provider id collisions), matching the complete tool_call. + if (!this.toolStarted.has(acc.toolCallId)) { + // Type boundary: before a new tool_call starts, flush any unclosed thinking/text + // segment. Only triggered on the tool's first delta (!toolStarted.has); continuation + // deltas (including id-less increments attributed via activeToolCallId) don't re-trigger the flush. + yield* this.flushThinking("completed"); + yield* this.flushText("completed"); + this.toolStarted.add(acc.toolCallId); + yield partialToolCall({ + eventType: "start", + name: acc.name, + toolCallId: acc.toolCallId, + }); + } + // Only emit a delta when there's an arguments increment (an empty increment carries + // no information, consistent with how empty text/thinking increments are handled). + // The delta doesn't repeat the name — leave it blank; tool identity is established by + // the start segment and tool_call_id. + if (item.arguments) { + acc.argsBuffer += item.arguments; + yield partialToolCall({ + eventType: "delta", + name: "", + arguments: item.arguments, + toolCallId: acc.toolCallId, + }); + } + break; + } + case "tool_call": { + // Complete tool-call content item: record the authoritative name/arguments and + // **eagerly emit** the tool's partial(stop) and complete tool_call. This lets the + // engine start approval/execution as soon as the first tool arrives, without waiting + // for the whole turn to finish — key for async/incremental tool calls (see comment + // #24). An empty id is invalid; skip it. + if (!item.tool_call_id) break; + // The model may think/output text before calling a tool: flush any buffered + // thinking/text complete messages before emitting the complete tool_call, so the + // complete-message order is thinking → text → tool_call. finish_reason isn't known + // yet here; the boundary flush uses completed, with the stop reason attributed to + // the tool_call itself. + yield* this.flushThinking("completed"); + yield* this.flushText("completed"); + // The same provider id already emitted a complete call and now another complete tool_call + // arrives: not a duplicate delivery but **another** call from a name-as-id provider (e.g. + // Gemini using the function name as id) — start a fresh accumulator, allocate a new unique id, + // and emit as usual; never drop it (otherwise parallel same-name calls in one turn would lack + // tool execution and paired output). + let acc = this.tools.get(item.tool_call_id); + if (!acc || acc.emitted) acc = this.createTool(item.tool_call_id, item.name); + acc.name = item.name; + acc.completeArgs = JSON.stringify(item.arguments ?? {}); + if (item.signature) acc.signature = item.signature; + yield* this.emitCompleteTool(acc); + break; + } + // Other content items (image_url / inline_data / inline_thinking / + // tool_result / embedding) are not treated as model streaming output. + default: + break; + } + } + } + + /** + * Stream-end finalization: first `yield` the `stop` and complete message for the text/thinking + * segments, then backfill partial(stop) + complete tool_call for tool_calls that **haven't + * been emitted eagerly yet** (i.e. the fallback case — no complete tool_call content item was + * received, only deltas). Tools already emitted eagerly in `pushEvent` are not repeated here. + */ + *finish(): Generator { + const stopReason = this.omniStopReason(); + + // 1. Close out the **last** unflushed thinking / text segment (earlier segments were already + // flushed at their respective type boundaries or before the first tool_call). + // Since boundaries already flush, at most one buffer is non-empty here, so call + // order doesn't matter — the other call is a no-op; the final finish_reason is used as + // the stop_reason here. + yield* this.flushThinking(stopReason); + yield* this.flushText(stopReason); + + // 2. Fallback path: for tools that never received a complete tool_call content item, + // backfill using the accumulated deltas. + for (const id of this.toolOrder) { + if (!id) continue; // Defensive: an invalid tool call with an empty id (its result can't be routed). + const acc = this.tools.get(id)!; + if (acc.emitted) continue; // Already emitted eagerly in pushEvent; don't repeat. + yield* this.emitCompleteTool(acc); + } + } + + /** + * Interruption finalization: even when interrupted or on error, close the structure + * as `start → delta → stop → complete message`. Closes any unclosed thinking/text segments and + * backfills their complete messages, then backfills partial(stop) + complete tool_call for + * tool_calls that only have deltas and were never emitted eagerly. All backfilled messages are + * uniformly tagged with the interruption `stopReason` (`aborted` / `timeout` / `failed`), to + * distinguish them from normal completion (`completed`). + * + * Differs from `finish`: doesn't read `finish_reason`, and doesn't produce `token_usage` (an + * interrupted Request has no usage to report); backfilled incomplete tool_calls carry the + * interruption stop_reason, so `context_engine` won't dispatch them for execution. + */ + *finishInterrupted(stopReason: StopReason): Generator { + yield* this.flushThinking(stopReason); + yield* this.flushText(stopReason); + for (const id of this.toolOrder) { + if (!id) continue; + const acc = this.tools.get(id)!; + if (acc.emitted) continue; // A tool_call already emitted eagerly keeps its `tool_call` semantics and is left unchanged. + yield* this.emitCompleteTool(acc, stopReason); + } + } + + /** + * Eagerly emits the finalization of a tool_call: partial(stop) (without name) + complete + * tool_call. Marks `emitted` to prevent duplicate emission in finish. `stopReason` defaults to + * `completed` (a normal request); when called during interruption finalization + * (`finishInterrupted`), the interruption reason is passed in, letting `context_engine` + * distinguish "a real tool request" from "an incomplete tool_call backfilled to close the + * structure on interruption" by stop_reason, and dispatch only the former. + */ + private *emitCompleteTool( + acc: ToolCallAccumulator, + stopReason: StopReason = "completed", + ): Generator { + acc.emitted = true; + if (this.toolStarted.has(acc.toolCallId)) { + // stop doesn't carry name (tool identity is established by start and tool_call_id). + yield partialToolCall({ + eventType: "stop", + name: "", + toolCallId: acc.toolCallId, + stopReason, + }); + } + yield toolCall({ + name: acc.name, + arguments: acc.completeArgs ?? acc.argsBuffer, + toolCallId: acc.toolCallId, + stopReason, + ...(acc.signature !== undefined ? { signature: acc.signature } : {}), + }); + // activeToolCallId holds the provider id key (used to attribute id-less deltas); reset it by providerKey. + if (this.activeToolCallId === acc.providerKey) { + this.activeToolCallId = null; + } + } + + /** + * Closes out the currently buffered thinking segment and appends the complete thinking + * message, then clears the buffer and resets the start flag. May be called before eagerly + * emitting the first complete tool_call (`pushEvent`) or at stream end (`finish`), which + * guarantees the complete-message order is thinking → text → tool_call. + * + * "Flush then reset" rather than a one-shot guard: once the buffer is cleared, a repeated call + * is a no-op (each segment is emitted exactly once); if the model outputs new thinking after a + * tool_call (interleaved/multi-segment models), that new segment accumulates again and gets + * correctly flushed at the next tool_call or `finish`, without being lost. + * + * The complete thinking message passes through the given stop_reason just like partial(stop) + * (aligned with flushText — streamed concatenation == complete message): at type boundaries the + * caller passes completed (the stop reason belongs to the following tool_call/text); + * finish/finishInterrupted follow the actual end reason. + */ + private *flushThinking(stopReason: StopReason): Generator { + if (this.thinkingStarted) { + yield partialThinking("stop", "", stopReason); + this.thinkingStarted = false; + } + // A thinking block with empty text but a signature (GPT-5 encrypted reasoning) still + // produces a complete message — the signature is required when replaying history. + if (this.thinkingBuffer || this.thinkingSignature !== undefined) { + yield thinkingMessage(this.thinkingBuffer, stopReason, { + ...(this.thinkingSignature !== undefined ? { signature: this.thinkingSignature } : {}), + }); + this.thinkingBuffer = ""; + this.thinkingSignature = undefined; + } + } + + /** + * Closes out the currently buffered text segment and appends the complete text message, then + * clears the buffer and resets the start flag. Uses the same "flush then reset" approach as + * `flushThinking` to support new text segments after a tool_call. When emitted before the + * first complete tool_call, finish_reason is unknown and the caller passes completed; when + * emitted in `finish`, the final `omniStopReason()` is passed (consistent with prior behavior). + */ + private *flushText(stopReason: StopReason): Generator { + if (this.textStarted) { + yield partialText("stop", "", stopReason); + this.textStarted = false; + } + // A text segment with empty text but a signature (e.g. Gemini carrying a thoughtSignature on + // a text part) still produces a complete message — aligned with flushThinking, so the + // signature isn't lost or leaked into a later segment just because the buffer is empty. + if (this.textBuffer || this.textSignature !== undefined) { + yield assistantText(this.textBuffer, stopReason, { + ...(this.textPhase != null ? { phase: this.textPhase } : {}), + ...(this.textSignature !== undefined ? { signature: this.textSignature } : {}), + }); + this.textBuffer = ""; + this.textSignature = undefined; + // textPhase is sticky across segments: a later segment with the same phase isn't + // re-marked; it's updated when a different phase marker appears. + } + } + + /** Whether finish_reason (the terminal event) has been received: signals a fully delivered response (see the defensive branch in streamGenerate). */ + sawFinishReason(): boolean { + return this.finishReason !== null; + } + + /** Token usage for this request (read after finish). */ + getRequestTokens(): TokenCounts { + return this.requestTokens; + } + + private ensureTool(providerKey: string, name: string): ToolCallAccumulator { + return this.tools.get(providerKey) ?? this.createTool(providerKey, name); + } + + /** + * Create an accumulator for a new call: the emitted id is made unique via the Session-level registry + * (kept as-is when the provider id is free, with a `#n` suffix on collision). Creating again under the + * same provider id (another call with a duplicate id) replaces the old Map entry — the old call has + * finished emitting, so later inbound events are attributed to the latest call. + */ + private createTool(providerKey: string, name: string): ToolCallAccumulator { + const acc: ToolCallAccumulator = { + name: name ?? "", + argsBuffer: "", + completeArgs: null, + toolCallId: this.toolCallIds.allocate(providerKey), + providerKey, + signature: undefined, + emitted: false, + }; + this.tools.set(providerKey, acc); + this.toolOrder.push(providerKey); + return acc; + } + + private usageOnce(usage: UsageMetadata): TokenCounts { + return usageToTokenCounts(usage); + } + + /** + * Converts an AgentHub finish_reason into an OmniMessage stop_reason (the five-value protocol). + * "stop", "tool_call", and null are treated as completed; length or unknown reasons map to failed. + */ + private omniStopReason(): StopReason { + if ( + this.finishReason == null || + this.finishReason === "stop" || + this.finishReason === "tool_call" + ) { + return "completed"; + } + return "failed"; + } +} + +/** + * One-shot translation: folds a batch of UniEvents into an OmniMessage sequence (including + * complete messages and token_usage). A pure function for easy unit testing; the live streaming + * path is wired up by `GenerativeModel.streamGenerate`. + * + * @param events The event sequence + * @param sessionTokensBefore The session's cumulative tokens before this translation (used to produce token_usage.session) + * @returns `{ messages, requestTokens, sessionTokens }` + */ +export function translateEvents( + events: UniEvent[], + sessionTokensBefore: TokenCounts = emptyTokenCounts(), +): { + messages: OmniMessage[]; + requestTokens: TokenCounts; + sessionTokens: TokenCounts; +} { + const translator = new EventTranslator(); + const out: OmniMessage[] = []; + for (const event of events) { + for (const msg of translator.pushEvent(event)) out.push(msg); + } + for (const msg of translator.finish()) out.push(msg); + + const requestTokens = translator.getRequestTokens(); + const sessionTokens = addTokenCounts(sessionTokensBefore, requestTokens); + out.push(tokenUsage(sessionTokens, requestTokens)); + + return { messages: out, requestTokens, sessionTokens }; +} + +// --------------------------------------------------------------------------- +// Retry policy +// --------------------------------------------------------------------------- + +/** + * Determines whether an error is an AgentHub / Provider response JSON parse error. + * + * AgentHub uses `JSON.parse` internally to parse response bodies, and a parse failure throws a + * `SyntaxError`; hence we judge directly by exception type (the `name` check also covers + * cross-realm or deserialization-reconstructed errors, and we probe down the `cause` chain for + * wrapped errors). This is not an auth/parameter failure but an incomplete LLM Request, and + * should end with `malformed` and be handed to the engine to retry. + */ +export function isMalformedJsonParseError(error: unknown): boolean { + if (error == null) return false; + if (error instanceof SyntaxError) return true; + const err = error as { name?: string; cause?: unknown }; + if (err.name === "SyntaxError") return true; + if (err.cause && err.cause !== error) { + return isMalformedJsonParseError(err.cause); + } + return false; +} + +/** + * Determines whether an error is AgentHub's "incomplete stream" validation error: when a + * server/proxy terminates the stream early **cleanly** at an event boundary (no network error + * thrown), AgentHub's `_validateLastEvent` reports the missing or incomplete final event as a + * plain `Error` ("Streaming response yielded no events" / "Last event must carry + * usage_metadata|finish_reason"). This is not an auth/parameter failure but an incomplete LLM + * Request, and should end with `malformed` and be handed to the engine to reconnect and retry. + * AgentHub doesn't provide an error type for this, so we match by message prefix + * (@prismshadow/agenthub 0.3.x), probing down the `cause` chain. + */ +export function isIncompleteStreamError(error: unknown): boolean { + if (error == null) return false; + const err = error as { message?: string; cause?: unknown }; + const msg = err.message ?? ""; + if ( + msg.startsWith("Streaming response yielded no events") || + msg.startsWith("Last event must carry") + ) { + return true; + } + if (err.cause && err.cause !== error) { + return isIncompleteStreamError(err.cause); + } + return false; +} + +/** + * Determines whether an error is retryable. + * + * Retryable: network errors, timeouts, connection reset, HTTP 429 / 5xx. + * Not retryable: HTTP 4xx auth/parameter errors (401/403/400/404, etc.). + * JSON parse errors are classified separately as `malformed` by `isMalformedJsonParseError`. + * + * Since AgentHub doesn't guarantee the shape of error objects, this uses a lenient check: first + * the status code, then error codes / message keywords. When undeterminable, treat it as + * **non-retryable** to avoid pointless retries. + */ +export function isRetryableError(error: unknown): boolean { + if (error == null) return false; + + const err = error as { + status?: number; + statusCode?: number; + code?: string; + name?: string; + message?: string; + }; + + // 1. HTTP status code takes priority. + const status = err.status ?? err.statusCode; + if (typeof status === "number") { + if (status === 429 || status === 408) return true; // Rate limited / request timeout (transient) + if (status >= 500 && status <= 599) return true; // Server error + if (status >= 400 && status <= 499) return false; // Other 4xx auth/parameter errors, not retryable + } + + // 2. Network-layer error codes. + const retryableCodes = new Set([ + "ECONNRESET", + "ECONNREFUSED", + "ETIMEDOUT", + "EPIPE", + "EAI_AGAIN", + "ENOTFOUND", + "ECONNABORTED", + ]); + if (err.code && retryableCodes.has(err.code)) return true; + + // 3. Error name / message keywords (timeout, network, rate limit). + if (err.name === "AbortError") return false; // User interruption, not retryable + const text = `${err.name ?? ""} ${err.message ?? ""}`.toLowerCase(); + if ( + /timeout|timed out|network|socket hang up|econnreset|connection reset|too many requests|rate limit|temporarily unavailable|503|502|504/.test( + text, + ) + ) { + return true; + } + + return false; +} + +// --------------------------------------------------------------------------- +// GenerativeModel +// --------------------------------------------------------------------------- + +/** + * A stateful LLM object attached to a Session. AgentHub's `streamingResponseStateful` maintains + * conversation history internally; this class is only responsible for protocol translation, + * streaming aggregation, and token accumulation. **It never retries internally** — retries are + * handled by `context_engine`. + */ +export class GenerativeModel implements LLMInterface { + private readonly client: AutoLLMClient; + private readonly uniConfig: UniConfig; + /** Streaming idle timeout (milliseconds); <= 0 disables it. A timeout is treated as needing reconnection. */ + private readonly requestTimeoutMs: number; + /** + * tool_call_id uniqueness registry (see tool-call-ids.ts): when a name-as-id provider (e.g. Gemini) + * calls the same tool repeatedly, it assigns a `#n` suffix to later calls so engine pairing and the + * frontend tool cards don't collide on id. Injected via config so it can be shared across the new + * instance rebuilt on compaction; defaults to a fresh one. + */ + private readonly toolCallIds: ToolCallIdAllocator; + + /** Cumulative session tokens. */ + sessionTokens: TokenCounts = emptyTokenCounts(); + + constructor(config: GenerativeModelConfig) { + // Omit apiKey / baseUrl when undefined, letting AgentHub read them from environment + // variables. clientType determines which protocol to speak (`openai` means OpenAI Chat + // Completions compatible); when omitted, AgentHub infers it from model_id, so it only needs + // to be specified explicitly for custom-named models. + this.client = new AutoLLMClient({ + model: config.modelId, + ...(config.apiKey !== undefined ? { apiKey: config.apiKey } : {}), + ...(config.baseUrl !== undefined ? { baseUrl: config.baseUrl } : {}), + ...(config.clientType !== undefined ? { clientType: config.clientType } : {}), + }); + + this.uniConfig = buildUniConfig(config); + this.requestTimeoutMs = config.requestTimeoutMs ?? 120000; + this.toolCallIds = config.toolCallIds ?? new ToolCallIdAllocator(); + } + + /** + * Streaming generation (a single attempt, no internal retry). Merges + * `params.newMessages` into one UniMessage to issue a stateful request, translating streamed + * UniEvents into OmniMessages. + * + * **Never throws to `context_engine`**: whether it ends normally or is interrupted/errors out, + * every `partial_*` segment is closed as `start → delta → stop → complete message`, + * and the terminal state is then returned as `LLMOutcome`: + * - **Normal completion**: `finish()` closes out and produces `token_usage` (usage is only + * produced in this case) → `completed`; + * - **Idle timeout / network drop** (retryable errors like network/429/5xx): + * `finishInterrupted("timeout")` closes out, produces no usage → `timeout`, reconnected by + * `context_engine` within the same run; + * - **AgentHub JSON parse error**: `finishInterrupted("malformed")` closes out, produces no + * usage → `malformed`, likewise reconnected by `context_engine` within the same run; + * - **User interruption**: `finishInterrupted("aborted")` closes out, produces no usage → + * `aborted`; + * - **Other non-retryable errors** (auth/parameters etc.): `finishInterrupted("failed")` + * closes out, produces no usage → `failed` (carrying `message`), handed to + * `context_engine` to stop and return control to the user. + * + * Timeout detection: the idle timer resets on every event received; once idle exceeds + * `requestTimeoutMs`, the underlying stream is aborted and handled as needing reconnection + * (merged with user interruption into a single internal AbortController). + */ + async *streamGenerate( + params: GenerativeModelParameters, + ): AsyncGenerator { + const userSignal = params.signal; + + // Already interrupted before issuing: no streaming segment has been opened, so nothing to close out. + if (userSignal?.aborted) return { status: "aborted" }; + + // Input merging is placed inside a guarded block: build failures such as empty input / + // mixed roles / argument JSON also collapse to a failed outcome, never throwing to + // context_engine. + let uniMessage: UniMessage; + try { + uniMessage = mergeOmniToUniMessage(params.newMessages); + } catch (err) { + return { + status: "failed", + message: err instanceof Error ? err.message : String(err), + }; + } + + const translator = new EventTranslator(this.toolCallIds); + + // Merges "user interruption" and "idle timeout" into a single internal AbortController: either triggering aborts the underlying stream. + const ac = new AbortController(); + const onUserAbort = (): void => ac.abort(); + userSignal?.addEventListener("abort", onUserAbort, { once: true }); + + let timedOut = false; + let timer: ReturnType | null = null; + const clearTimer = (): void => { + if (timer) { + clearTimeout(timer); + timer = null; + } + }; + const armTimer = (): void => { + if (this.requestTimeoutMs <= 0) return; // Timeout disabled + clearTimer(); + timer = setTimeout(() => { + timedOut = true; + ac.abort(); + }, this.requestTimeoutMs); + }; + + // Terminal-state classification: timeout (timed out/network drop) / malformed (response + // parse error) / aborted (user) / failed (other). null means it ended normally. + let outcome: LLMOutcome | null = null; + try { + const it = this.openStream(uniMessage, ac.signal)[Symbol.asyncIterator](); + for (;;) { + // The interruption check must happen **before pulling from upstream**: the user may + // interrupt while this generator is suspended at the `yield` below (the typical case — + // the engine is blocked on `await approve(tc)` waiting for human approval). By then, + // onUserAbort has already called `ac.abort()`, cutting off the upstream stream; when the + // consumer pulls again and we come back here to call `it.next()` on an **already-aborted + // stream**, that promise will never settle. The idle timer can't save us either: once it + // fires, it just calls `ac.abort()` again (already aborted, a no-op), and the pending + // `it.next()` still hangs forever. The consequence is that `run` never closes out and the + // Session stays stuck running forever — after interruption, it can neither send messages + // nor compact. + if (userSignal?.aborted) { + outcome = { status: "aborted" }; + break; + } + // Timing runs **only while waiting on an upstream event** (excluding consumer/yield + // time), measuring upstream idleness — this avoids a slow consumer (e.g. a slow Trace + // sink) falsely triggering the timeout. + armTimer(); + let res: IteratorResult; + try { + res = await it.next(); + } finally { + clearTimer(); + } + if (res.done) break; + if (userSignal?.aborted) { + outcome = { status: "aborted" }; + break; + } + for (const msg of translator.pushEvent(res.value)) yield msg; + } + } catch (error) { + // User interruption **takes priority**: even if the idle timer fires at the same time, + // it's classified as aborted (user intent outweighs a coincidental timeout). + if (userSignal?.aborted) { + outcome = { status: "aborted" }; + } else if (timedOut) { + outcome = { status: "timeout" }; // Idle timeout -> needs reconnection + } else if (isMalformedJsonParseError(error) || isIncompleteStreamError(error)) { + // A response JSON parse error, or a cleanly truncated stream (AgentHub's final-event + // validation failed): both are an incomplete LLM Request, handed to the engine as + // malformed to reconnect and retry — must not be classified as failed. + outcome = { + status: "malformed", + message: error instanceof Error ? error.message : String(error), + }; + } else if (isRetryableError(error)) { + outcome = { status: "timeout" }; // Network drop/network error -> needs reconnection + } else if ((error as { name?: string })?.name === "AbortError") { + outcome = { status: "aborted" }; // Fallback: an unexpected abort (neither timeout nor user) + } else { + outcome = { + status: "failed", + message: error instanceof Error ? error.message : String(error), + }; + } + } finally { + clearTimer(); + userSignal?.removeEventListener("abort", onUserAbort); + } + + // Defensive: when the underlying stream responds to an abort with a **graceful end (done)** + // rather than throwing, it must still be closed out as interrupted/timed out, and must not be + // misjudged as completed (priority matches the catch classification: user interruption > + // timeout). Exception: if finish_reason was already received before the stream ended (the + // response was fully delivered and AgentHub has already committed this turn into stateful + // history), close out as completed — if the interruption race lands exactly during the wait + // on the final next() and gets misjudged as aborted, the already-committed tool_use turn + // would be cleaned up by context_engine as "incomplete" flatten, losing the tool_result + // pairing; subsequent requests would all be rejected by the provider as an unanswered + // tool_use (400), and the engine has no fix-up path left that touches LLM history. + if (!outcome && !translator.sawFinishReason()) { + if (userSignal?.aborted) outcome = { status: "aborted" }; + else if (timedOut) outcome = { status: "timeout" }; + } + + if (outcome) { + // Interrupted/errored: close any opened streaming segments and backfill the complete message, producing no token_usage. + const reason: StopReason = outcome.status === "completed" ? "failed" : outcome.status; + for (const msg of translator.finishInterrupted(reason)) yield msg; + return outcome; + } + + // Normal completion: backfill stop + the complete model_msg, and produce token_usage. + for (const msg of translator.finish()) yield msg; + const requestTokens = translator.getRequestTokens(); + this.sessionTokens = addTokenCounts(this.sessionTokens, requestTokens); + yield tokenUsage(this.sessionTokens, requestTokens); + return { status: "completed" }; + } + + /** + * Injects the replayed history in one shot when resuming a Session: converts the complete + * OmniMessage history, grouped by adjacent same role, into + * AgentHub UniMessages and calls AgentHub's setHistory, so subsequent Requests continue from a + * history exactly matching the original conversation. **Called only once, on a fresh context + * object, during resumption**; not used during normal operation, where the incremental context + * is maintained by AgentHub itself. + * Docs: /docs/sessions-and-traces § "Session recovery". + */ + setHistory(history: OmniMessage[]): void { + if (history.length === 0) return; + // Resume seeding: register tool_call_ids already used in history into the uniqueness registry. A + // name-as-id provider (e.g. Gemini) only gets a new suffix when it calls the same tool again after + // resume, so it won't collide with the history tool cards the frontend already rendered. + for (const msg of history) { + const p = msg.payload as { type?: string; tool_call_id?: string }; + if (p.type === "tool_call" && p.tool_call_id) { + this.toolCallIds.markUsed(p.tool_call_id); + } + } + this.client.setHistory(groupHistoryToUniMessages(history)); + } + + /** + * Opens the underlying AgentHub stream (a testing seam): defaults to + * `streamingResponseStateful`; unit tests can subclass and override this method, feeding in a + * controlled UniEvent stream to verify the outcome classification for timeout/network + * drop/interruption/error (without a real API). + */ + protected openStream(uniMessage: UniMessage, signal: AbortSignal): AsyncIterable { + return this.client.streamingResponseStateful({ + message: uniMessage, + config: this.uniConfig, + signal, + }); + } +} + +// --------------------------------------------------------------------------- +// UniConfig pre-construction +// --------------------------------------------------------------------------- + +/** Maps a ThinkingLevelName to the AgentHub ThinkingLevel enum; returns undefined if not found. */ +export function mapThinkingLevel(name: ThinkingLevelName | undefined): ThinkingLevel | undefined { + if (name === undefined) return undefined; + const table: Record = { + none: ThinkingLevel.NONE, + low: ThinkingLevel.LOW, + medium: ThinkingLevel.MEDIUM, + high: ThinkingLevel.HIGH, + xhigh: ThinkingLevel.XHIGH, + }; + return table[name]; +} + +/** Maps ToolDefinition[] to AgentHub ToolSchema[]. */ +export function toolDefinitionsToSchemas(tools: ToolDefinition[]): ToolSchema[] { + return tools.map((tool) => ({ + name: tool.name, + description: tool.description, + ...(tool.parameters !== undefined ? { parameters: tool.parameters } : {}), + })); +} + +/** Pre-builds UniConfig from GenerativeModelConfig (called once at construction time). */ +export function buildUniConfig(config: GenerativeModelConfig): UniConfig { + const uniConfig: UniConfig = { + tools: toolDefinitionsToSchemas(config.tools), + }; + if (config.systemPrompt !== undefined) { + uniConfig.system_prompt = config.systemPrompt; + } + if (config.maxTokens !== undefined) { + uniConfig.max_tokens = config.maxTokens; + } + const thinking = mapThinkingLevel(config.thinkingLevel); + if (thinking !== undefined) { + uniConfig.thinking_level = thinking; + } + return uniConfig; +} diff --git a/packages/core/src/llm/index.ts b/packages/core/src/llm/index.ts new file mode 100644 index 0000000..c4cbd03 --- /dev/null +++ b/packages/core/src/llm/index.ts @@ -0,0 +1,22 @@ +/** + * LLM interface module entry point. + * + * Exports `GenerativeModel` (the LLMInterface implementation), along with internal + * pure conversion functions for unit testing (message merging, event translation, + * token accounting, UniConfig construction, retry determination). + */ +export { + GenerativeModel, + EventTranslator, + groupHistoryToUniMessages, + mergeOmniToUniMessage, + translateEvents, + usageToTokenCounts, + isMalformedJsonParseError, + isIncompleteStreamError, + isRetryableError, + mapThinkingLevel, + toolDefinitionsToSchemas, + buildUniConfig, +} from "./generative-model.js"; +export { ToolCallIdAllocator, stripToolCallIdSuffix } from "./tool-call-ids.js"; diff --git a/packages/core/src/llm/tool-call-ids.ts b/packages/core/src/llm/tool-call-ids.ts new file mode 100644 index 0000000..69ffe67 --- /dev/null +++ b/packages/core/src/llm/tool-call-ids.ts @@ -0,0 +1,51 @@ +/** + * Session-level uniqueness for tool_call_id. + * + * Some providers don't produce a real call id: e.g. Gemini's functionCall has no id, so AgentHub + * uses the **function name** as the `tool_call_id` — consecutive/parallel calls to the same tool then + * all share one id. But the OmniMessage world (engine dispatch/pairing, approval routing, frontend + * tool-card attribution) keys on `tool_call_id`, and a collision lets a later call overwrite the + * earlier one (parallel same-name calls in one turn can even be dropped entirely). + * + * Approach: inbound, `EventTranslator` disambiguates duplicate ids with a `#n` suffix (the first keeps + * the original id); outbound (returning tool_result, replaying history on resume) uses + * `stripToolCallIdSuffix` to strip the suffix and restore the original — Gemini's functionResponse + * pairs by using `tool_call_id` as the name, so it must be restored to the function name. The registry + * lives at Session level (the new GenerativeModel rebuilt on compaction shares the same instance), and + * on resume `setHistory` seeds it with historical ids, so the uniqueness scope covers the entire + * context the frontend renders. + * Docs: /docs/interfaces § "The built-in implementation: GenerativeModel". + */ +export class ToolCallIdAllocator { + /** OmniMessage-level tool_call_ids already taken in this Session (history-seeded + allocated). */ + private used = new Set(); + + /** Register an already-used id (for resume seeding); registering twice is harmless. */ + markUsed(id: string): void { + this.used.add(id); + } + + /** + * Allocate a Session-unique id for a provider-reported tool_call_id: if unused, keep the original; + * if already used (a repeat call from a name-as-id provider), take the first free `origId#n` (n from 2). + * Providers with truly unique ids (OpenAI `call_*` / Claude `toolu_*`) never collide, so they pass through unchanged. + */ + allocate(providerId: string): string { + let id = providerId; + for (let n = 2; this.used.has(id); n += 1) { + id = `${providerId}#${n}`; + } + this.used.add(id); + return id; + } +} + +/** + * Strip the `#n` suffix added by `allocate`, restoring the provider's original id (returns as-is when + * there's no suffix; idempotent). On resume there's no registry to compare against, so it trims by + * shape: real ids from known providers (OpenAI/Claude `call_*`/`toolu_*`, Gemini function names — `#` + * isn't a valid function-name char) never end in `#`, so they aren't harmed. + */ +export function stripToolCallIdSuffix(id: string): string { + return id.replace(/#\d+$/, ""); +} diff --git a/packages/core/src/omnimessage/aggregate.ts b/packages/core/src/omnimessage/aggregate.ts new file mode 100644 index 0000000..862f0df --- /dev/null +++ b/packages/core/src/omnimessage/aggregate.ts @@ -0,0 +1,155 @@ +/** + * Aggregates streaming partial_* messages into a complete model_msg. + * + * When recording a Trace, streaming `partial_*` messages must first be joined into a complete + * `model_msg` before writing. This module provides: + * - `PartialAggregator`: a stateful aggregator, pushed one message at a time, producing a + * complete message when a fragment ends with `stop`; + * - `aggregateAll`: a one-shot pass that collapses `partial_*` messages in an array into + * complete messages. + * + * Complete / event / session_meta messages pass through unchanged, preserving their original + * order. + * Docs: /docs/omni-message § "The streaming discipline". + */ +import { assistantText, thinkingMessage, toolCall, toolCallOutput } from "./builders.js"; +import type { OmniMessage, PartialModelPayload, StopReason } from "./types.js"; +import { isPartialPayload } from "./types.js"; + +type PartialKind = PartialModelPayload["type"]; + +interface OpenFragment { + kind: PartialKind; + /** Accumulation buffer for text / thinking / tool_call arguments / tool_call_output. */ + buffer: string; + name?: string; + toolCallId?: string; + /** Images carried by tool_call_output (images aren't incremental — a single delta carries the whole set; a later one overwrites). */ + images?: string[]; + lastStopReason: StopReason; +} + +/** Merge key for partial fragments: same type + same tool_call_id counts as the same fragment. */ +function fragmentKey(p: PartialModelPayload): string { + const id = "tool_call_id" in p ? p.tool_call_id : ""; + return `${p.type}::${id}`; +} + +function finalize(frag: OpenFragment): OmniMessage { + switch (frag.kind) { + case "partial_text": + return assistantText(frag.buffer, frag.lastStopReason); + case "partial_thinking": + return thinkingMessage(frag.buffer, frag.lastStopReason); + case "partial_tool_call": + return toolCall({ + name: frag.name ?? "", + arguments: frag.buffer, + toolCallId: frag.toolCallId ?? "", + stopReason: frag.lastStopReason, + }); + case "partial_tool_call_output": + return toolCallOutput({ + output: frag.buffer, + toolCallId: frag.toolCallId ?? "", + stopReason: frag.lastStopReason, + ...(frag.images !== undefined ? { images: frag.images } : {}), + }); + } +} + +function appendDelta(frag: OpenFragment, p: PartialModelPayload): void { + switch (p.type) { + case "partial_text": + frag.buffer += p.text; + break; + case "partial_thinking": + frag.buffer += p.thinking; + break; + case "partial_tool_call": + frag.buffer += p.arguments; + if (p.name) frag.name = p.name; + frag.toolCallId = p.tool_call_id; + break; + case "partial_tool_call_output": + frag.buffer += p.output; + if (p.images && p.images.length > 0) frag.images = p.images; + frag.toolCallId = p.tool_call_id; + break; + } + if (p.stop_reason !== undefined) frag.lastStopReason = p.stop_reason; +} + +function newFragment(p: PartialModelPayload): OpenFragment { + const frag: OpenFragment = { + kind: p.type, + buffer: "", + lastStopReason: "completed", + }; + if (p.type === "partial_tool_call") { + frag.name = p.name; + frag.toolCallId = p.tool_call_id; + } else if (p.type === "partial_tool_call_output") { + frag.toolCallId = p.tool_call_id; + } + return frag; +} + +/** + * Stateful aggregator. Pushed one message at a time via `push`: + * - complete / event / session_meta messages are returned unchanged; + * - `partial_*` messages accumulate into an internal fragment, producing a complete message + * when `event_type === "stop"`; + * - `flush` forcibly emits any fragments that haven't yet received a stop. + */ +export class PartialAggregator { + private open = new Map(); + + push(msg: OmniMessage): OmniMessage[] { + if (!isPartialPayload(msg.payload)) { + return [msg]; + } + const p = msg.payload; + const key = fragmentKey(p); + let frag = this.open.get(key); + + if (p.event_type === "start") { + // start reopens a fragment; if a fragment with the same key already exists (out-of-order), finalize it first. + const out: OmniMessage[] = []; + if (frag) out.push(finalize(frag)); + frag = newFragment(p); + appendDelta(frag, p); + this.open.set(key, frag); + return out; + } + + if (!frag) { + // delta/stop without a preceding start: handle leniently, creating a new fragment as needed. + frag = newFragment(p); + this.open.set(key, frag); + } + appendDelta(frag, p); + + if (p.event_type === "stop") { + this.open.delete(key); + return [finalize(frag)]; + } + return []; + } + + /** Finalizes: emits all still-open fragments (in the order they were opened). */ + flush(): OmniMessage[] { + const out = [...this.open.values()].map(finalize); + this.open.clear(); + return out; + } +} + +/** One-shot aggregation: keeps non-partial messages in their original order, collapsing partial ones into complete messages. */ +export function aggregateAll(messages: OmniMessage[]): OmniMessage[] { + const agg = new PartialAggregator(); + const out: OmniMessage[] = []; + for (const msg of messages) out.push(...agg.push(msg)); + out.push(...agg.flush()); + return out; +} diff --git a/packages/core/src/omnimessage/builders.ts b/packages/core/src/omnimessage/builders.ts new file mode 100644 index 0000000..4311f42 --- /dev/null +++ b/packages/core/src/omnimessage/builders.ts @@ -0,0 +1,342 @@ +/** + * OmniMessage builders. All modules create messages exclusively through these builders, avoiding + * ad hoc protocol structures scattered across the codebase. + * Every builder writes an ISO 8601 UTC timestamp. + * Docs: /docs/omni-message § "Builders and guards". + */ +import type { + AbortPayload, + ApprovalDecision, + ApprovalDecisionPayload, + CompactionBeginPayload, + CompactionEndPayload, + CompactionMode, + CompactionReason, + EventMessage, + ImageUrlPayload, + InlineDataPayload, + InlineThinkingPayload, + MessageOrigin, + ModelMessage, + OmniMessage, + PartialTextPayload, + PartialThinkingPayload, + PartialToolCallOutputPayload, + PartialToolCallPayload, + RequestBeginPayload, + RequestEndPayload, + Role, + SessionMetaMessage, + SessionMetaPayload, + StopReason, + StreamEventType, + SubagentPayload, + TextPayload, + ThinkingPayload, + TokenCounts, + TokenUsagePayload, + ToolCallOutputPayload, + ToolCallPayload, +} from "./types.js"; + +/** The current moment's ISO 8601 UTC timestamp. */ +function nowIso(): string { + return new Date().toISOString(); +} + +function model

(payload: P): OmniMessage

{ + return { timestamp: nowIso(), type: "model_msg", payload }; +} + +function event

(payload: P): OmniMessage

{ + return { timestamp: nowIso(), type: "event_msg", payload }; +} + +// session_meta --------------------------------------------------------------- + +export function sessionMeta(payload: SessionMetaPayload): SessionMetaMessage { + return { timestamp: nowIso(), type: "session_meta", payload }; +} + +// Complete model_msg ----------------------------------------------------------- + +/** + * Provider fidelity fields: kept as-is and restored verbatim on replay. + * Builder convention: positional-argument-style builders carry these in a trailing `fidelity` + * object (narrowed via Pick per payload type — e.g. thinking only has signature); object-argument- + * style builders (toolCall) flatten `fidelity` fields into the parameter object alongside + * `stopReason`, mirroring the payload structure directly. + */ +export interface FidelityFields { + phase?: string | null; + signature?: string; +} + +export function textMessage( + role: Role, + text: string, + stopReason: StopReason = "completed", + fidelity?: FidelityFields, +): OmniMessage { + return model({ + type: "text", + role, + text, + stop_reason: stopReason, + ...(fidelity?.phase != null ? { phase: fidelity.phase } : {}), + ...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}), + }); +} + +export const userText = (text: string): OmniMessage => textMessage("user", text); + +export const assistantText = ( + text: string, + stopReason: StopReason = "completed", + fidelity?: FidelityFields, +): OmniMessage => textMessage("assistant", text, stopReason, fidelity); + +export function imageUrlMessage(imageUrl: string): OmniMessage { + return model({ + type: "image_url", + role: "user", + image_url: imageUrl, + stop_reason: "completed", + }); +} + +export function inlineData( + role: Role, + data: string, + mimeType: string, + fidelity?: Pick, +): OmniMessage { + return model({ + type: "inline_data", + role, + data, + mime_type: mimeType, + stop_reason: "completed", + ...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}), + }); +} + +export function thinkingMessage( + thinking: string, + stopReason: StopReason = "completed", + fidelity?: Pick, +): OmniMessage { + return model({ + type: "thinking", + role: "assistant", + thinking, + stop_reason: stopReason, + ...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}), + }); +} + +export function inlineThinking( + data: string, + mimeType: string, + fidelity?: Pick, +): OmniMessage { + return model({ + type: "inline_thinking", + role: "assistant", + data, + mime_type: mimeType, + stop_reason: "completed", + ...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}), + }); +} + +export function toolCall(args: { + name: string; + arguments: string; + toolCallId: string; + stopReason?: StopReason; + signature?: string; +}): OmniMessage { + return model({ + type: "tool_call", + role: "assistant", + name: args.name, + arguments: args.arguments, + tool_call_id: args.toolCallId, + stop_reason: args.stopReason ?? "completed", + ...(args.signature !== undefined ? { signature: args.signature } : {}), + }); +} + +export function toolCallOutput(args: { + output: string; + toolCallId: string; + stopReason?: StopReason; + /** Images carried by the tool output (array of data URLs); images aren't incremental — a single delta carries the whole set in the streaming path, and the complete message carries them too. */ + images?: string[]; +}): OmniMessage { + return model({ + type: "tool_call_output", + role: "user", + output: args.output, + ...(args.images !== undefined && args.images.length > 0 ? { images: args.images } : {}), + tool_call_id: args.toolCallId, + stop_reason: args.stopReason ?? "completed", + }); +} + +// Streaming partial_* model_msg ------------------------------------------------- + +export function partialText( + eventType: StreamEventType, + text = "", + stopReason: StopReason = "completed", +): OmniMessage { + return model({ + type: "partial_text", + role: "assistant", + event_type: eventType, + text, + stop_reason: stopReason, + }); +} + +export function partialThinking( + eventType: StreamEventType, + thinking = "", + stopReason: StopReason = "completed", +): OmniMessage { + return model({ + type: "partial_thinking", + role: "assistant", + event_type: eventType, + thinking, + stop_reason: stopReason, + }); +} + +export function partialToolCall(args: { + eventType: StreamEventType; + name: string; + arguments?: string; + toolCallId: string; + stopReason?: StopReason; +}): OmniMessage { + return model({ + type: "partial_tool_call", + role: "assistant", + event_type: args.eventType, + name: args.name, + arguments: args.arguments ?? "", + tool_call_id: args.toolCallId, + stop_reason: args.stopReason ?? "completed", + }); +} + +export function partialToolCallOutput(args: { + eventType: StreamEventType; + output?: string; + toolCallId: string; + stopReason?: StopReason; + /** Images carried by the tool output (array of data URLs); images aren't incremental — a single delta carries the whole set. */ + images?: string[]; +}): OmniMessage { + return model({ + type: "partial_tool_call_output", + role: "user", + event_type: args.eventType, + output: args.output ?? "", + ...(args.images !== undefined && args.images.length > 0 ? { images: args.images } : {}), + tool_call_id: args.toolCallId, + stop_reason: args.stopReason ?? "completed", + }); +} + +// event_msg ------------------------------------------------------------------- + +export function approvalDecision( + decision: ApprovalDecision, + toolCallId: string, +): OmniMessage { + return event({ type: "approval_decision", decision, tool_call_id: toolCallId }); +} + +export function abortEvent(reason: string | null = null): OmniMessage { + return event({ type: "abort", reason }); +} + +/** request begin event: marks the start of one LLM Request. */ +export function requestBegin(): OmniMessage { + return event({ type: "request_begin" }); +} + +/** request end event: carries the terminal state (`completed` means this turn was already committed to AgentHub). */ +export function requestEnd(status: StopReason): OmniMessage { + return event({ type: "request_end", status }); +} + +/** compaction begin event: carries the trigger reason, mode, current context usage, and cumulative Session turn count. */ +export function compactionBegin(args: { + reason: CompactionReason; + mode: CompactionMode; + context: number; + turns: number; +}): OmniMessage { + return event({ + type: "compaction_begin", + reason: args.reason, + mode: args.mode, + context: args.context, + turns: args.turns, + }); +} + +/** compaction end event: carries the compaction result (non-`completed` means compaction was abandoned and the original context is kept). */ +export function compactionEnd(args: { + reason: CompactionReason; + mode: CompactionMode; + status: StopReason; +}): OmniMessage { + return event({ + type: "compaction_end", + reason: args.reason, + mode: args.mode, + status: args.status, + }); +} + +/** subagent derivation pointer event: records only the direct child session's Session id (written to the parent Trace by context_engine). */ +export function subagentEvent(sessionId: string): OmniMessage { + return event({ type: "subagent", session_id: sessionId }); +} + +export function emptyTokenCounts(): TokenCounts { + return { cache_read: 0, cache_write: 0, output: 0, total: 0 }; +} + +export function tokenUsage( + session: TokenCounts, + request: TokenCounts, +): OmniMessage { + return event({ type: "token_usage", session, request }); +} + +/** Adds two sets of Token counts together, used to maintain cumulative Session usage. */ +export function addTokenCounts(a: TokenCounts, b: TokenCounts): TokenCounts { + return { + cache_read: a.cache_read + b.cache_read, + cache_write: a.cache_write + b.cache_write, + output: a.output + b.output, + total: a.total + b.total, + }; +} + +/** + * Marks a message with a nested-origin tag: prepends one hop (a child Session id) to the front + * of `origin`, outer-to-inner. + * Used by host tools (e.g. run_subagent) when forwarding child-session messages; an absent + * `origin` means the message comes from the main Session. + */ +export function withOrigin(msg: M, sessionId: MessageOrigin): M { + return { ...msg, origin: [sessionId, ...(msg.origin ?? [])] }; +} diff --git a/packages/core/src/omnimessage/index.ts b/packages/core/src/omnimessage/index.ts new file mode 100644 index 0000000..ee2d6f7 --- /dev/null +++ b/packages/core/src/omnimessage/index.ts @@ -0,0 +1,3 @@ +export * from "./types.js"; +export * from "./builders.js"; +export * from "./aggregate.js"; diff --git a/packages/core/src/omnimessage/types.ts b/packages/core/src/omnimessage/types.ts new file mode 100644 index 0000000..07f56e1 --- /dev/null +++ b/packages/core/src/omnimessage/types.ts @@ -0,0 +1,400 @@ +/** + * OmniMessage — PenguinHarness's primary message protocol. + * + * All messages share one envelope: `timestamp` (ISO 8601 UTC), `type`, and `payload`. + * The outer `type` falls into three categories: + * - `session_meta`: Session metadata; + * - `model_msg`: model input/output messages (both complete messages and streaming + * `partial_*` messages); + * - `event_msg`: control/statistics events during execution. + * + * Trace records only: `session_meta`, complete `model_msg`, and all `event_msg`; + * the Human interface communicates using: complete `model_msg`, streaming `partial_*`, and all + * `event_msg`. + * + * Docs: packages/docs/content/omni-message.{zh,en}.md (site path /docs/omni-message) documents + * this protocol payload-for-payload — keep the page in sync when changing types here. + */ + +/** The outer message category. */ +export type OmniMessageType = "session_meta" | "model_msg" | "event_msg"; + +/** The message's originating role. */ +export type Role = "user" | "assistant"; + +/** + * The reason a model response or message generation ended. Only five protocol values are + * allowed: + * - `completed`: finished normally, including completed text, thinking, tool requests, or + * tool output; + * - `failed`: a non-retryable error or tool execution failure; + * - `aborted`: user-initiated interruption or cancellation; + * - `timeout`: LLM request timed out; + * - `malformed`: the LLM response was malformed (e.g. AgentHub JSON parsing exception). + * Only LLM timeout / malformed trigger a context_engine reconnect. + * Docs: /docs/omni-message § "stop_reason". + */ +export type StopReason = "completed" | "failed" | "aborted" | "timeout" | "malformed"; + +/** The event phase of a streaming fragment. `stop` marks the end of a fragment and usually carries no incremental content. */ +export type StreamEventType = "start" | "delta" | "stop"; + +/** + * Nested-origin marker: a child Session id. The message envelope's `origin` is a chain of child + * Session ids ordered **outer-to-inner**, identifying that the message comes from a nested child + * session (e.g. a child Session derived by `run_subagent`); each layer of host-tool forwarding + * prepends one more hop at the front. **An absent `origin` (the message carries no `origin`) + * means the message comes from the main Session itself** (an empty array is never produced + * either). Only session_id is recorded: the corresponding tool_call / agent info can be obtained + * from the `run_subagent` tool_call in the parent session's stream and the child Session's own + * Trace (session_meta). + * Docs: /docs/omni-message § "origin: the Subagent chain". + */ +export type MessageOrigin = string; + +/** The approval decision for a tool call. */ +export type ApprovalDecision = "allow" | "deny"; + +/** Token counts (input/output/cache/total). */ +export interface TokenCounts { + cache_read: number; + cache_write: number; + output: number; + total: number; +} + +// --------------------------------------------------------------------------- +// session_meta +// --------------------------------------------------------------------------- +// Docs: /docs/omni-message § "session_meta" + +/** Tool definition passed to the LLM (OpenAI/JSON Schema style). */ +export interface ToolDefinition { + name: string; + description: string; + parameters?: Record; +} + +export interface SessionMetaPayload { + session_id: string; + /** The session model's provider group (paired with `model_id` to form a model reference). */ + provider: string; + /** The session model's upstream model_id (the request id sent to AgentHub; paired with `provider`). */ + model_id: string; + model_context_window: number | string; + /** The system prompt actually used by this Session (the assembled result with environment placeholders already substituted). */ + system_prompt: string; + /** The list of tool definitions this Session exposes to the model (full schema, matching what's sent to the LLM). */ + tools: ToolDefinition[]; + /** The model's thinking level (from system_config.model.thinking_level; "default" when unconfigured). */ + thinking_level: string; + /** Absolute path to the Agent State. */ + agent_state: string; + /** Absolute path to the Workspace. */ + workspace: string; +} + +// --------------------------------------------------------------------------- +// model_msg — complete messages +// --------------------------------------------------------------------------- +// Docs: /docs/omni-message § "model_msg: complete payloads" + +export interface TextPayload { + type: "text"; + role: Role; + text: string; + stop_reason?: StopReason; + /** Provider fidelity field: text phase marker (e.g. GPT-5 segments by phase), kept as-is and restored verbatim. */ + phase?: string | null; + /** Provider fidelity field: signature, kept as-is and restored verbatim. */ + signature?: string; +} + +export interface ImageUrlPayload { + type: "image_url"; + role: "user"; + /** A web URL or a base64 data URL. */ + image_url: string; + stop_reason?: StopReason; +} + +export interface InlineDataPayload { + type: "inline_data"; + role: Role; + /** Base64-encoded bytes. */ + data: string; + mime_type: string; + stop_reason?: StopReason; + /** Provider fidelity field: signature, kept as-is and restored verbatim. */ + signature?: string; +} + +export interface ThinkingPayload { + type: "thinking"; + role: "assistant"; + thinking: string; + stop_reason?: StopReason; + /** + * Provider fidelity field: thinking-block signature (Claude thinking blocks / redacted + * thinking, GPT-5 encrypted reasoning, etc. — **required** when some models replay history), + * kept as-is and restored verbatim — losing it breaks Session resumption. + */ + signature?: string; +} + +export interface InlineThinkingPayload { + type: "inline_thinking"; + role: "assistant"; + /** Base64-encoded bytes. */ + data: string; + mime_type: string; + stop_reason?: StopReason; + /** Provider fidelity field: signature, kept as-is and restored verbatim. */ + signature?: string; +} + +export interface ToolCallPayload { + type: "tool_call"; + role: "assistant"; + name: string; + /** Tool arguments as a JSON string. */ + arguments: string; + tool_call_id: string; + stop_reason?: StopReason; + /** Provider fidelity field: signature, kept as-is and restored verbatim. */ + signature?: string; +} + +export interface ToolCallOutputPayload { + type: "tool_call_output"; + role: "user"; + output: string; + /** + * Images carried by the tool output (optional): each is a `data:;base64,...` data URL, + * fed back to the model alongside the text (e.g. images read by read_image). Images aren't + * incremental: the streaming path carries the whole set once via a single delta (see + * `PartialToolCallOutputPayload.images`), and the complete message carries them again — the + * streamed-and-joined result equals the complete message. + */ + images?: string[]; + tool_call_id: string; + stop_reason?: StopReason; +} + +// --------------------------------------------------------------------------- +// model_msg — streaming partial_* messages +// --------------------------------------------------------------------------- +// Docs: /docs/omni-message § "model_msg: streaming partials" + +export interface PartialTextPayload { + type: "partial_text"; + role: "assistant"; + event_type: StreamEventType; + text: string; + stop_reason?: StopReason; +} + +export interface PartialThinkingPayload { + type: "partial_thinking"; + role: "assistant"; + event_type: StreamEventType; + thinking: string; + stop_reason?: StopReason; +} + +export interface PartialToolCallPayload { + type: "partial_tool_call"; + role: "assistant"; + event_type: StreamEventType; + name: string; + /** Incremental fragment of the arguments JSON. */ + arguments: string; + tool_call_id: string; + stop_reason?: StopReason; +} + +export interface PartialToolCallOutputPayload { + type: "partial_tool_call_output"; + role: "user"; + event_type: StreamEventType; + output: string; + /** Images carried by the tool output (optional): images aren't incremental, carried as a whole by a single delta (consistent with the complete message). */ + images?: string[]; + tool_call_id: string; + stop_reason?: StopReason; +} + +// --------------------------------------------------------------------------- +// event_msg +// --------------------------------------------------------------------------- +// Docs: /docs/omni-message § "event_msg" + +export interface ApprovalDecisionPayload { + type: "approval_decision"; + decision: ApprovalDecision; + tool_call_id: string; +} + +export interface AbortPayload { + type: "abort"; + reason?: string | null; +} + +export interface TokenUsagePayload { + type: "token_usage"; + /** Current Session cumulative token usage. */ + session: TokenCounts; + /** Token usage for the most recent Request. */ + request: TokenCounts; +} + +/** + * Request boundary event: the boundary of one LLM Request, produced **in pairs** by + * `context_engine` and written to Trace. `request_end` + * with `status` of `completed` means the turn has been committed by AgentHub — this is the + * mechanical criterion Trace replay (Session resumption) uses to determine whether a turn was + * committed, and it also gives performance analysis a basis for Request latency and turn counts. + * A compaction request produces this same event pair too (written to Trace only, not streamed). + */ +export interface RequestBeginPayload { + type: "request_begin"; +} + +export interface RequestEndPayload { + type: "request_end"; + /** Terminal state of this Request (reuses the five StopReason values, sharing its source with this turn's complete message's stop_reason / LLMOutcome). */ + status: StopReason; +} + +/** Compaction trigger reason: context threshold / turn-count threshold / user-initiated request. */ +export type CompactionReason = "context" | "turns" | "manual"; + +/** Context compaction mode: summary relay / direct discard. */ +export type CompactionMode = "summarize" | "discard"; + +/** + * Compaction boundary event: the compaction process exposes + * only this event pair to Human, produced **in pairs** by `context_engine`. Both `reason` and + * `mode` are carried on both events, for stateless frontend rendering; `status` reuses the + * five-value `StopReason` protocol (compaction converges to a terminal state, taking + * `completed` / `failed` / `aborted` in practice — `timeout` / `malformed` are handled internally + * by the compaction request's existing retry mechanism, collapsing to `failed` once retries are + * exhausted). + */ +export interface CompactionBeginPayload { + type: "compaction_begin"; + reason: CompactionReason; + mode: CompactionMode; + /** Current context token usage (the most recent token_usage's request.total). */ + context: number; + /** Session cumulative turn count. */ + turns: number; +} + +export interface CompactionEndPayload { + type: "compaction_end"; + reason: CompactionReason; + mode: CompactionMode; + /** Compaction result; non-`completed` means compaction was abandoned and the original context was kept. */ + status: StopReason; +} + +/** + * Subagent pointer event: when the parent Session spawns a + * **direct** child session, `context_engine` writes this to the parent Trace (not streamed), + * recording only the child session's Session id — the child session's other details live in its + * own Trace's `session_meta`. When the session is reopened, the server uses this to recursively + * expand the child Trace and reconstruct the `origin` chain; a grandchild session's pointer is + * recorded by the child Trace itself. + */ +export interface SubagentPayload { + type: "subagent"; + /** The direct child session's Session id. */ + session_id: string; +} + +// --------------------------------------------------------------------------- +// Union types and the message envelope +// --------------------------------------------------------------------------- + +/** Complete model_msg payload (written to Trace and exposed externally). */ +export type CompleteModelPayload = + | TextPayload + | ImageUrlPayload + | InlineDataPayload + | ThinkingPayload + | InlineThinkingPayload + | ToolCallPayload + | ToolCallOutputPayload; + +/** Streaming model_msg payload. */ +export type PartialModelPayload = + | PartialTextPayload + | PartialThinkingPayload + | PartialToolCallPayload + | PartialToolCallOutputPayload; + +export type ModelPayload = CompleteModelPayload | PartialModelPayload; + +export type EventPayload = + | ApprovalDecisionPayload + | AbortPayload + | RequestBeginPayload + | RequestEndPayload + | TokenUsagePayload + | CompactionBeginPayload + | CompactionEndPayload + | SubagentPayload; + +export type OmniPayload = SessionMetaPayload | ModelPayload | EventPayload; + +/** The unified message envelope. */ +export interface OmniMessage

{ + /** ISO 8601 UTC timestamp. */ + timestamp: string; + type: OmniMessageType; + payload: P; + /** Nested-origin marker: the chain of child Session ids ordered outer-to-inner; absent = from the main Session (see MessageOrigin). */ + origin?: MessageOrigin[]; +} + +// Convenience aliases for concrete message types -------------------------------- + +export type SessionMetaMessage = OmniMessage; +export type ModelMessage = OmniMessage; +export type EventMessage = OmniMessage; +export type CompleteModelMessage = OmniMessage; +export type PartialModelMessage = OmniMessage; + +// --------------------------------------------------------------------------- +// Runtime discrimination helpers +// --------------------------------------------------------------------------- + +/** The set of type values for streaming partial_* payloads. */ +const PARTIAL_PAYLOAD_TYPES = [ + "partial_text", + "partial_thinking", + "partial_tool_call", + "partial_tool_call_output", +] as const; + +export function isPartialPayload(p: OmniPayload): p is PartialModelPayload { + return (PARTIAL_PAYLOAD_TYPES as readonly string[]).includes((p as { type?: string }).type ?? ""); +} + +export function isModelMessage(msg: OmniMessage): msg is ModelMessage { + return msg.type === "model_msg"; +} + +export function isEventMessage(msg: OmniMessage): msg is EventMessage { + return msg.type === "event_msg"; +} + +export function isSessionMeta(msg: OmniMessage): msg is SessionMetaMessage { + return msg.type === "session_meta"; +} + +/** A complete model_msg (not partial_*), i.e. a message that can be written to Trace. */ +export function isCompleteModelMessage(msg: OmniMessage): msg is CompleteModelMessage { + return msg.type === "model_msg" && !isPartialPayload(msg.payload); +} diff --git a/packages/core/src/session-title.ts b/packages/core/src/session-title.ts new file mode 100644 index 0000000..f17d88c --- /dev/null +++ b/packages/core/src/session-title.ts @@ -0,0 +1,124 @@ +/** + * Session title generation: an **out-of-band, one-off request** that generates + * a short title from the first-turn conversation text. + * + * Called by `session.generateTitle()`: sends one request using the bare LLM for the session's + * Model (no tools, no system prompt, thinking off), without writing history or Trace. Material + * defaults to what the Session self-captures during run (see session.ts); this module is only + * responsible for the prompt format, driving the one-off request, and sanitizing the result — + * when to generate a title and where to store it is decided by the host (Web server / CLI). + */ +import { userText } from "./omnimessage/index.js"; +import type { + OmniMessage, + TextPayload, + TokenCounts, + TokenUsagePayload, +} from "./omnimessage/index.js"; +import type { LLMInterface } from "./interfaces.js"; + +/** Cap on conversation text spliced into the title request (user/model each truncated separately, to control cost). */ +const EXCERPT_MAX_CHARS = 2000; +/** Cap on title length (fallback truncation for when the model occasionally ignores the constraint). */ +const TITLE_MAX_CHARS = 30; + +export interface SessionTitleResult { + /** The sanitized title; null when material is insufficient, the request fails, or the output is empty. */ + title: string | null; + /** Token consumption for this request (accumulated token_usage.request); null if no request occurred or there's no usage. */ + usage: TokenCounts | null; +} + +/** + * Assembles the title-generation Prompt (exported for host/test assertion use). Uses English + * instructions to avoid polluting the title's language, and requires the **title to be in the + * same language as the conversation** (English conversation gets an English title, Chinese + * conversation gets a Chinese title); when assistant material is empty, it relies on the user + * request alone. + */ +export function buildTitlePrompt(userExcerpt: string, assistantExcerpt: string): string { + const clip = (s: string) => (s.length > EXCERPT_MAX_CHARS ? s.slice(0, EXCERPT_MAX_CHARS) : s); + const lines = [ + "Generate a concise title for the conversation below.", + "Rules:", + "- Write the title in the SAME language the user is using.", + "- Keep it short: at most 6 words, or ~16 characters for CJK.", + "- Output ONLY the title text — no quotes, no trailing punctuation, no explanation.", + "", + "[User]", + clip(userExcerpt), + ]; + if (assistantExcerpt.trim()) { + lines.push("", "[Assistant]", clip(assistantExcerpt)); + } + return lines.join("\n"); +} + +/** Sanitizes model output into a title: strips leading/trailing quotes/brackets and trailing punctuation (until stable), collapses whitespace, and truncates if too long; returns null for an empty result. */ +export function sanitizeTitle(raw: string): string | null { + let t = raw.replace(/\s+/g, " ").trim(); + // Stripping quotes can expose more punctuation underneath (or vice versa), so strip repeatedly until stable. + for (let prev = ""; prev !== t;) { + prev = t; + t = t + .replace(/^["'“”‘’「」『』《》〈〉【】()()\s]+/, "") + .replace(/["'“”‘’「」『』《》〈〉【】()()\s]+$/, "") + .replace(/[。..!!??;;,,、::]+$/, "") + .trim(); + } + if (!t) return null; + return t.length > TITLE_MAX_CHARS ? t.slice(0, TITLE_MAX_CHARS) : t; +} + +/** + * Drives a single title-generation request: collects model text and token_usage, and resolves + * based on the outcome. Generation only requires user material (assistant material may be + * empty — a pure tool-only turn can still get a title); no request is sent if user material is + * empty; `title` is null if the request doesn't complete (any usage already produced is still + * returned). + */ +export async function generateTitleWithLLM( + llm: LLMInterface, + args: { userText: string; assistantText: string; signal?: AbortSignal }, +): Promise { + if (!args.userText.trim()) { + return { title: null, usage: null }; + } + const prompt = buildTitlePrompt(args.userText, args.assistantText); + const gen = llm.streamGenerate({ + newMessages: [userText(prompt)], + ...(args.signal ? { signal: args.signal } : {}), + }); + let collected = ""; + let usage: TokenCounts | null = null; + for (;;) { + const step = await gen.next(); + if (step.done) { + if (step.value.status !== "completed") return { title: null, usage }; + break; + } + const msg = step.value; + if (isAssistantText(msg)) collected += msg.payload.text; + if (isTokenUsage(msg)) { + const r = msg.payload.request; + usage = usage + ? { + cache_read: usage.cache_read + r.cache_read, + cache_write: usage.cache_write + r.cache_write, + output: usage.output + r.output, + total: usage.total + r.total, + } + : { ...r }; + } + } + return { title: sanitizeTitle(collected), usage }; +} + +function isAssistantText(msg: OmniMessage): msg is OmniMessage { + const payload = msg.payload as { type?: string; role?: string }; + return msg.type === "model_msg" && payload.type === "text" && payload.role === "assistant"; +} + +function isTokenUsage(msg: OmniMessage): msg is OmniMessage { + return msg.type === "event_msg" && (msg.payload as { type?: string }).type === "token_usage"; +} diff --git a/packages/core/src/session.ts b/packages/core/src/session.ts new file mode 100644 index 0000000..1558a18 --- /dev/null +++ b/packages/core/src/session.ts @@ -0,0 +1,250 @@ +/** + * Session — a continuous conversation context under the same Agent and Workspace. + * + * Human is the SDK's input/output boundary: there is no "Human + * implementation/interface". + * - Input: the OmniMessage list (Prompt) passed to `run(newMessages, opts?)`, plus the abort + * signal `signal` and the per-call approval callback `approve` in `opts`; + * - Output: `run` streams OmniMessage via an async generator. + * + * Approval is a **within-turn interaction**: as soon as a tool_call finishes streaming, `approve` + * is requested immediately, and it executes if allowed. Approvals for multiple tools happen one + * at a time, but execution doesn't block the generation/approval of subsequent tools (execution + * can overlap). GenerativeModel maintains history across turns/Tasks. A Task ends when a turn no + * longer produces a tool_call (final reply). + * + * Rendering tool calls is not Session/core's responsibility: the CLI / Web frontend renders it + * from the streamed OmniMessage on its own. + * Docs: /docs/agent-loop; /docs/interfaces § "The Human boundary". + */ +import { sessionMeta } from "./omnimessage/index.js"; +import type { OmniMessage, SessionMetaPayload, TokenCounts } from "./omnimessage/index.js"; +import { imagesToScratchpadPaths } from "./internal/session-support.js"; +import type { EnvironmentInterface, LLMInterface, ToolPermission } from "./interfaces.js"; +import { generateTitleWithLLM } from "./session-title.js"; +import type { SessionTitleResult } from "./session-title.js"; +import { ContextEngine } from "./engine/context-engine.js"; +import type { + CompactAvailability, + CompactionSettings, + EngineInitialState, + RunOptions, + TraceSink, +} from "./engine/context-engine.js"; + +export interface SessionConfig { + /** Session metadata (session_id / provider / model_id / model_context_window / system_prompt / tools / thinking_level / agent_state / workspace). */ + meta: SessionMetaPayload; + llm: LLMInterface; + environment: EnvironmentInterface; + trace?: TraceSink; + maxTurns?: number; + /** Creates a new LLM object after compaction (carries over the Session's accumulated Token count); context compaction is unavailable if not provided. */ + createLLM?: (sessionTokens: TokenCounts) => LLMInterface; + /** + * Factory for the bare LLM used by out-of-band, one-off requests (same Model/credential as + * the session; no tools, no system prompt, thinking off): used for meta-requests such as + * `generateTitle`; if not provided, `generateTitle` returns null. + */ + createBareLLM?: () => LLMInterface; + /** Context compaction settings (defaults are filled in by the composition layer); only takes effect when provided together with `createLLM`. */ + compaction?: CompactionSettings; + /** Session resume: `session_meta` is already in the original Trace file, so it isn't written again on the first run (avoids duplication). */ + metaAlreadyWritten?: boolean; + /** Session resume: the engine's initial state derived from Trace replay (carry-over / accumulated stats, etc.). */ + initialEngineState?: EngineInitialState; + /** Session resume: the full historical messages of the current context (for rendering, including interrupted turns and their markers), for frontend display. */ + resumedHistory?: OmniMessage[]; + /** + * Set when the session's model doesn't support images (the composition layer decides this via + * ModelEntry.vision): images in `run` input are saved to this directory (the session's + * scratchpad), and the path is appended to the user text instead — the model views the image + * via describe_image, and images never enter the session history directly (some providers + * return a 400 outright on image input). + */ + inputImagesDir?: string; +} + +/** Cap on captured title material (chars per side, matching buildTitlePrompt's truncation); stops accumulating once exceeded. */ +const TITLE_MATERIAL_LIMIT = 2000; + +/** + * Accumulates title material: the body text of complete text messages from the main session + * (no origin) — thinking and tool calls naturally don't count — and stops once the cap is hit. + */ +function appendTitleText(base: string, msg: OmniMessage, role: "user" | "assistant"): string { + if (base.length >= TITLE_MATERIAL_LIMIT) return base; + if (msg.origin && msg.origin.length > 0) return base; + const p = msg.payload as { type?: string; role?: string; text?: string }; + if (msg.type !== "model_msg" || p.type !== "text" || p.role !== role || !p.text) return base; + return base ? `${base}\n${p.text}` : p.text; +} + +export class Session { + readonly sessionId: string; + /** The session model's provider group (paired with `modelId` to form the model reference). */ + readonly provider: string; + /** The session model's upstream model_id (the request id sent to AgentHub). */ + readonly modelId: string; + readonly workspaceDir: string; + /** Session resume: the full historical messages of the current context (for rendering); undefined for a non-resumed Session. */ + readonly resumedHistory?: OmniMessage[]; + + private readonly engine: ContextEngine; + private readonly environment: EnvironmentInterface; + private readonly trace?: TraceSink; + private readonly meta: OmniMessage; + private readonly createBareLLM?: () => LLMInterface; + private readonly inputImagesDir?: string; + private metaWritten = false; + /** Title material (used by `generateTitle` as the default): the user input and model body text of the first Task that contains user text. */ + private titleUserText = ""; + private titleAssistantText = ""; + /** Material-frozen flag: becomes true once the first Task containing user text finishes; subsequent runs stop accumulating. */ + private titleMaterialFrozen = false; + + constructor(config: SessionConfig) { + this.sessionId = config.meta.session_id; + this.provider = config.meta.provider; + this.modelId = config.meta.model_id; + this.workspaceDir = config.meta.workspace; + this.environment = config.environment; + this.trace = config.trace; + this.meta = sessionMeta(config.meta); + this.metaWritten = config.metaAlreadyWritten ?? false; + if (config.resumedHistory) this.resumedHistory = config.resumedHistory; + if (config.createBareLLM) this.createBareLLM = config.createBareLLM; + if (config.inputImagesDir) this.inputImagesDir = config.inputImagesDir; + this.engine = new ContextEngine({ + llm: config.llm, + environment: config.environment, + ...(config.trace ? { trace: config.trace } : {}), + ...(config.maxTurns !== undefined ? { maxTurns: config.maxTurns } : {}), + // Context compaction: new LLM factory + resolved settings + writes session_meta at the start of the new Trace file after splitting. + ...(config.createLLM ? { createLLM: config.createLLM } : {}), + ...(config.compaction ? { compaction: config.compaction } : {}), + ...(config.initialEngineState ? { initialState: config.initialEngineState } : {}), + sessionMeta: this.meta, + }); + } + + /** + * Runs a Task to completion and streams out OmniMessage. `newMessages` is this call's Prompt + * (only the newly added input); `opts` carries the abort signal `signal` and the per-call + * approval callback `approve` (the engine calls it once per tool_call within a turn). + * On the first run, `session_meta` is written to the Trace first. + * + * A single `run` automatically drives the whole ReAct loop: consuming the LLM stream, + * approving and executing tools one at a time, feeding results back for the next turn, + * until a turn no longer produces a tool_call (Task ends) or it's aborted. + * Docs: /docs/agent-loop § "The loop at a glance". + */ + async *run(newMessages: OmniMessage[], opts?: RunOptions): AsyncGenerator { + // Model doesn't support images: input images are saved to disk first (session scratchpad), + // then the path is appended to the text before it reaches the engine/Trace. + if (this.inputImagesDir) { + newMessages = await imagesToScratchpadPaths(newMessages, this.inputImagesDir); + } + await this.ensureMetaWritten(); + // Self-captures title material (the title is derived from the first-turn + // conversation text): while material isn't frozen yet, collect this call's user text and + // the produced model text; freezes once the first Task containing user text finishes, so + // the title reflects the start of the conversation. + const capture = !this.titleMaterialFrozen; + if (capture) { + for (const m of newMessages) { + this.titleUserText = appendTitleText(this.titleUserText, m, "user"); + } + } + for await (const msg of this.engine.run(newMessages, opts)) { + if (capture) { + this.titleAssistantText = appendTitleText(this.titleAssistantText, msg, "assistant"); + } + yield msg; + } + if (capture && this.titleUserText.trim()) this.titleMaterialFrozen = true; + } + + /** + * User-initiated request to compact context (e.g. a CLI command): reuses the automatic + * compaction flow but skips the threshold check (reason=manual). Only callable at Task + * boundaries (between runs); streams out paired `compaction` events. The summarize digest + * becomes the prefix of the next `run`'s input (merged with the next user Prompt). A no-op + * if compaction isn't configured. + * Docs: /docs/agent-loop § "Compaction". + */ + async *compact(opts?: { signal?: AbortSignal }): AsyncGenerator { + yield* this.engine.compact(opts); + } + + /** + * Whether compaction is possible, and why not if not (see ContextEngine.compactability). + * When the result isn't `ok`, `compact()` is a no-op and yields no messages — callers should + * give feedback based on this rather than triggering a silent, fruitless compaction. + */ + compactability(): CompactAvailability { + return this.engine.compactability(); + } + + /** Writes `session_meta` to the Trace before the first run/compaction; best-effort — failure doesn't interrupt the run. */ + private async ensureMetaWritten(): Promise { + if (this.metaWritten) return; + if (this.trace) { + try { + await this.trace.write(this.meta); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + process.stderr.write(`[trace] session_meta write failed: ${message}\n`); + } + } + this.metaWritten = true; + } + + /** + * Out-of-band, one-off request that generates a short title from the first-turn conversation + * text: sends one request using the bare LLM for the session's Model (no + * tools, no system prompt, thinking off), **without writing history or Trace**. Material + * defaults to the first Task text self-captured by the Session (user input and model body + * text collected during run; thinking and tool calls don't count), so callers don't need to + * supply it; `material` can override this (e.g. when a host generates a title for a + * sub-session — the material is that sub-session's own conversation). `title` is null if the + * material is empty, the request fails, or the composition layer didn't supply a bare LLM + * factory. Token consumption is returned via `usage` for the host to account for. + * Docs: /docs/agent-loop § "Side channels". + */ + async generateTitle(args?: { + /** Material override; defaults to the Session's self-captured material. */ + material?: { userText: string; assistantText: string }; + signal?: AbortSignal; + }): Promise { + if (!this.createBareLLM) return { title: null, usage: null }; + const material = args?.material ?? { + userText: this.titleUserText, + assistantText: this.titleAssistantText, + }; + return generateTitleWithLLM(this.createBareLLM(), { + ...material, + ...(args?.signal ? { signal: args.signal } : {}), + }); + } + + /** Queries a tool's permission level (for the frontend to determine permission mode); returns undefined for unknown tools. */ + toolPermission(name: string): ToolPermission | undefined { + return this.environment.toolPermission(name); + } + + /** This Session's session_meta message (used e.g. by host tools to forward nested-session metadata to a parent session). */ + get metaMessage(): OmniMessage { + return this.meta; + } + + /** + * Releases runtime resources held by the Session: kills long-running command sessions + * managed by the Environment. The host calls this when the Session ends (CLI exit, Web + * session close) to avoid leaking background processes into the host process's lifetime. + * Optional, idempotent. + */ + dispose(): void { + this.environment.dispose?.(); + } +} diff --git a/packages/core/src/state/agent-state.ts b/packages/core/src/state/agent-state.ts new file mode 100644 index 0000000..0f9ef09 --- /dev/null +++ b/packages/core/src/state/agent-state.ts @@ -0,0 +1,418 @@ +/** + * Loading and initialization of Agent State (semantics modeled on Hugging Face model loading). + * + * - Initializes when the target Agent directory is empty (no `system_config.yaml`): creates + * `agent_state/`, `tools/`, `memory/`, `skills/`, and the sibling `scratchpad/`, and writes + * the default `system_config.yaml` and `AGENTS.md`. + * - Otherwise loads the existing system config and editable Prompt for the given `agentId`. + * + * The full runtime Prompt is rendered from the system-level Prompt template in + * `system_config.yaml`; placeholders in the template are replaced with `AGENTS.md` and the + * concrete Session runtime environment fields. Built-in tools and MCP Server config + * come from `system_config.yaml`. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { parse as parseYaml, stringify as stringifyYaml } from "yaml"; +import { + loadLibrarySkills, + parseSkillFrontmatter, + type SkillMetadata, +} from "@prismshadow/penguin-skills"; +import type { ToolConfig, ToolDefinitionConfig } from "../interfaces.js"; +import { + AGENT_ID_PLACEHOLDER, + AGENTS_MD_PLACEHOLDER, + VAULT_KEYS_PLACEHOLDER, + SKILL_METADATA_PLACEHOLDER, + CWD_PLACEHOLDER, + DATE_PLACEHOLDER, + defaultAgentsMd, + defaultSystemConfig, + OS_VERSION_PLACEHOLDER, + PLATFORM_PLACEHOLDER, + PROJECT_DIR_PLACEHOLDER, + SESSION_ID_PLACEHOLDER, + type SystemConfig, +} from "./default-config.js"; +import { builtinProjectAgentPresets, type AgentPreset } from "./builtin-agents.js"; +import { provisionExampleBenchmark } from "./example-benchmark.js"; +import { + agentsMdPath, + agentStateDir, + DEFAULT_AGENT_ID, + DEFAULT_PROJECT_ID, + memoryDir, + resolveRoot, + scratchpadDir, + skillsDir, + systemConfigPath, + toolsDir, +} from "./paths.js"; + +/** project_id / agent_id / skill_name only allow letters, digits, underscore `_`, and hyphen `-` (prevents path traversal). */ +const ID_PATTERN = /^[A-Za-z0-9_-]+$/; +export type IdKind = "project_id" | "agent_id" | "skill_name"; + +export function isValidId(id: string): boolean { + return ID_PATTERN.test(id); +} + +export function assertValidId(kind: IdKind, id: string): void { + if (!ID_PATTERN.test(id)) { + throw new Error( + `Invalid ${kind} ${JSON.stringify(id)}: only letters, digits, "_" and "-" are allowed.`, + ); + } +} + +/** A loaded Agent State handle. */ +export interface AgentState { + root: string; + projectId: string; + agentId: string; + stateDir: string; + systemConfig: SystemConfig; + agentsMd: string; +} + +export interface SessionEnvironmentValues { + sessionId: string; + cwd: string; + /** The Agent id this Session belongs to (system Prompt placeholder {{AGENT_ID}}). */ + agentId: string; + /** Absolute path to this Project's directory (system Prompt placeholder {{PROJECT_DIR}}; Agent State/scratchpad paths are derived from it). */ + projectDir: string; + platform: string; + osVersion: string; + date: string; +} + +/** + * Loads or initializes Agent State. + * + * When root/project/agent are omitted, `resolveRoot()` and the default constants are used. If + * `system_config.yaml` doesn't exist, the directory is treated as empty and initialized; + * otherwise the existing content is loaded. `preset` only takes effect on the initialization + * path (name/description/AGENTS.md overrides and extra Skills) and is ignored when loading an + * existing Agent — existing config is never overwritten. + */ +export async function loadOrInitAgentState(opts?: { + agentId?: string; + projectId?: string; + root?: string; + preset?: AgentPreset; +}): Promise { + const root = opts?.root ?? resolveRoot(); + const projectId = opts?.projectId ?? DEFAULT_PROJECT_ID; + const agentId = opts?.agentId ?? DEFAULT_AGENT_ID; + + // Validate before building paths, to prevent path traversal. + assertValidId("project_id", projectId); + assertValidId("agent_id", agentId); + + const stateDir = agentStateDir(root, projectId, agentId); + const configPath = systemConfigPath(root, projectId, agentId); + const mdPath = agentsMdPath(root, projectId, agentId); + + let systemConfig: SystemConfig; + let agentsMd: string; + + if (await fileExists(configPath)) { + // Load path: read the existing system_config.yaml and AGENTS.md. + const rawConfig = await fs.readFile(configPath, "utf8"); + const parsed = parseYaml(rawConfig) as unknown; + // Defensive check: if the file is empty/corrupted, parseYaml may return null/a non-object, + // or system_prompt may be missing — otherwise "undefined" would get spliced into the system + // Prompt. Throw a clear error when validation fails. + if ( + parsed === null || + typeof parsed !== "object" || + typeof (parsed as SystemConfig).system_prompt !== "string" + ) { + throw new Error(`Agent State 配置非法:${configPath} 为空、损坏或缺少 system_prompt 字段。`); + } + systemConfig = parsed as SystemConfig; + agentsMd = (await fileExists(mdPath)) ? await fs.readFile(mdPath, "utf8") : defaultAgentsMd(); + } else { + // Init path: create the directory structure and write default config (preset only takes effect here). + await Promise.all([ + fs.mkdir(stateDir, { recursive: true }), + fs.mkdir(toolsDir(root, projectId, agentId), { recursive: true }), + fs.mkdir(memoryDir(root, projectId, agentId), { recursive: true }), + fs.mkdir(skillsDir(root, projectId, agentId), { recursive: true }), + fs.mkdir(scratchpadDir(root, projectId, agentId), { recursive: true }), + ]); + const preset = opts?.preset; + systemConfig = { + ...defaultSystemConfig(), + ...(preset?.name !== undefined ? { name: preset.name } : {}), + ...(preset?.description !== undefined ? { description: preset.description } : {}), + }; + agentsMd = preset?.agentsMd ?? defaultAgentsMd(); + // Only installs the Skills specified by preset (a plain newly created Agent gets none + // pre-installed). A default_agent with no + // preset (e.g. created on first CLI run) still gets every Skill in the library pre-installed + // — the install policy follows Agent identity, not whether creation came from the server or + // was done directly via SDK/CLI. + // Skills have no dedicated tool: metadata is injected via {{SKILL_METADATA}}, and the model + // reads SKILL.md with shell and follows it. + const skills = + opts?.preset === undefined && agentId === DEFAULT_AGENT_ID + ? loadLibrarySkills() + : (opts?.preset?.skills ?? []); + await Promise.all([ + fs.writeFile(mdPath, agentsMd, "utf8"), + ...skills.map((skill) => installSkill(root, projectId, agentId, skill)), + // The example Benchmark is only provisioned alongside default_agent (so the evaluation + // center has data out of the box): idempotently skipped if benchmarks/ already exists, + // and not created for plain Agents. + ...(agentId === DEFAULT_AGENT_ID + ? [provisionExampleBenchmark(root, projectId, agentId)] + : []), + ]); + // system_config.yaml is written last: its existence is the "initialization complete" marker + // (the load/init decision point). If this fails partway (disk full / crash), the next run + // still takes the init path and self-heals, so no half-initialized state with missing Skills is left behind. + await fs.writeFile(configPath, stringifyYaml(systemConfig), "utf8"); + } + + return { root, projectId, agentId, stateDir, systemConfig, agentsMd }; +} + +/** + * Initializes a Project's built-in Agent (the only built-in Agent: default_agent). + * + * Calls loadOrInitAgentState for each one: an Agent whose directory already exists (including a + * default_agent created earlier by the CLI) is only loaded, never overwritten (preset only + * takes effect on initialization). Returns the list of built-in Agent ids. + */ +export async function provisionProjectAgents(opts?: { + root?: string; + projectId?: string; +}): Promise { + const agentIds: string[] = []; + for (const { agentId, preset } of builtinProjectAgentPresets()) { + await loadOrInitAgentState({ + ...(opts?.root !== undefined ? { root: opts.root } : {}), + ...(opts?.projectId !== undefined ? { projectId: opts.projectId } : {}), + agentId, + preset, + }); + agentIds.push(agentId); + } + return agentIds; +} + +/** + * The vault key-name list: the replacement value for `{{VAULT_KEYS}}`, one `- KEY` per line; + * returns an empty string when there are no keys. + * **Contains only key names, never values** — values are only injected into the exec_command + * subprocess environment, never the model context. The statement of the vault's purpose is part + * of the default template body (the # Vault section) and is kept even with no vault. + */ +function vaultKeysList(keys: string[]): string { + return keys.map((key) => `- ${key}`).join("\n"); +} + +/** + * Installs a Skill into the target Agent: writes `skills//SKILL.md` verbatim (the full + * SKILL.md content including frontmatter, ensuring a trailing newline); if the directory + * already exists, it's overwritten (reinstalling = updating to the latest content). An optional + * icon.svg is written alongside SKILL.md; if this install doesn't + * include an icon, any old icon.svg is removed, preserving "overwrite update" semantics (the + * directory content matches the Skill being installed). + * Docs: /docs/skills § "Installation and storage". + */ +export async function installSkill( + root: string, + projectId: string, + agentId: string, + skill: { name: string; content: string; icon?: string }, +): Promise { + assertValidId("project_id", projectId); + assertValidId("agent_id", agentId); + assertValidId("skill_name", skill.name); + const dir = path.join(skillsDir(root, projectId, agentId), skill.name); + await fs.mkdir(dir, { recursive: true }); + const content = skill.content.endsWith("\n") ? skill.content : `${skill.content}\n`; + const iconPath = path.join(dir, "icon.svg"); + await Promise.all([ + fs.writeFile(path.join(dir, "SKILL.md"), content, "utf8"), + skill.icon !== undefined + ? fs.writeFile(iconPath, skill.icon, "utf8") + : fs.rm(iconPath, { force: true }), + ]); +} + +/** Uninstalls a Skill: deletes the entire `skills//` directory; idempotent, no error if it doesn't exist. */ +export async function removeSkill( + root: string, + projectId: string, + agentId: string, + name: string, +): Promise { + assertValidId("project_id", projectId); + assertValidId("agent_id", agentId); + assertValidId("skill_name", name); + await fs.rm(path.join(skillsDir(root, projectId, agentId), name), { + recursive: true, + force: true, + }); +} + +/** An installed Skill entry: frontmatter metadata (including an optional short description) + the optional icon.svg content in the directory. */ +export interface InstalledSkill extends SkillMetadata { + /** The raw content of `skills//icon.svg` (a custom icon copied alongside SKILL.md at install time); the field is omitted when missing (the frontend falls back to a default book icon). */ + icon?: string; +} + +/** + * Lists the metadata of Skills installed on the target Agent: scans `skills//SKILL.md` and + * parses its frontmatter (optional fields like short_description(_zh) pass through as parsed), + * also reading the optional icon.svg content in the directory. Tolerant: a directory whose + * frontmatter fails to parse or is missing `name` falls back to + * `{ name: , description: "", version: 1, updated: "" }`; a directory with no + * SKILL.md doesn't count as a Skill; returns [] if skills/ doesn't exist. Results are sorted by + * name (a stable order for both Prompt injection and API responses). + * Docs: /docs/skills § "Installation and storage". + */ +export async function listInstalledSkills( + root: string, + projectId: string, + agentId: string, +): Promise { + assertValidId("project_id", projectId); + assertValidId("agent_id", agentId); + const dir = skillsDir(root, projectId, agentId); + let entries; + try { + entries = await fs.readdir(dir, { withFileTypes: true }); + } catch { + return []; + } + const skills: InstalledSkill[] = []; + for (const entry of entries) { + if (!entry.isDirectory()) continue; + let raw: string; + try { + raw = await fs.readFile(path.join(dir, entry.name, "SKILL.md"), "utf8"); + } catch { + continue; + } + let icon: string | undefined; + try { + icon = await fs.readFile(path.join(dir, entry.name, "icon.svg"), "utf8"); + } catch { + // icon.svg is optional: missing means no custom icon. + } + // The directory name is the Skill's identity (install / uninstall / Prompt read guidance all + // address by directory name): frontmatter only supplies display fields like description; when + // its `name` doesn't match the directory name (a hand-written or network-sourced Skill), the + // directory name always wins — otherwise the model would read a nonexistent path using the + // injected name, and the API couldn't uninstall it either. + const parsed = parseSkillFrontmatter(raw); + skills.push({ + ...(parsed ?? { description: "", version: 1, updated: "" }), + name: entry.name, + ...(icon !== undefined ? { icon } : {}), + }); + } + return skills.sort((a, b) => a.name.localeCompare(b.name)); +} + +/** + * Skill metadata section: the replacement value for `{{SKILL_METADATA}}`, one line per Skill in + * the form `- \`name\` — description` (just the name when description is empty); an empty array + * returns an empty string. The full body is read by the model on demand via shell. + */ +export function skillMetadataSection(skills: SkillMetadata[]): string { + return skills + .map((s) => (s.description ? `- \`${s.name}\` — ${s.description}` : `- \`${s.name}\``)) + .join("\n"); +} + +/** + * Renders the complete runtime system Prompt: substitutes `AGENTS.md`, vault key names, Skill + * metadata, and the concrete Session runtime environment placeholders into the system Prompt + * template. The assembly layer only does placeholder substitution and adds no extra text — + * wrapper text such as `` and the # Vault / # Skills statements are + * written directly into the system Prompt template itself (the Prompt is fully + * transparent and editable via `system_config.yaml`). Other files in Agent State / Workspace are + * never auto-injected. + * + * `{{VAULT_KEYS}}` is replaced with the vault key-name list (an empty string if empty/not + * provided): this lets the model know which APIs requiring a key it can call; values are never + * injected. `{{SKILL_METADATA}}` is replaced with the installed Skills' metadata lines (an empty + * string if empty/not provided). A custom template that removes a placeholder gets no + * corresponding content injected. + * Docs: /docs/configuration § "System prompt placeholders". + */ +export function assembleSystemPrompt( + state: AgentState, + sessionEnvironment?: SessionEnvironmentValues, + vaultKeys?: string[], + skillMetadata?: SkillMetadata[], +): string { + return state.systemConfig.system_prompt + .split(AGENTS_MD_PLACEHOLDER) + .join(state.agentsMd.trim()) + .split(VAULT_KEYS_PLACEHOLDER) + .join(vaultKeysList(vaultKeys ?? [])) + .split(SKILL_METADATA_PLACEHOLDER) + .join(skillMetadataSection(skillMetadata ?? [])) + .split(AGENT_ID_PLACEHOLDER) + .join(sessionEnvironment?.agentId ?? state.agentId) + .split(PROJECT_DIR_PLACEHOLDER) + .join(sessionEnvironment?.projectDir ?? "") + .split(SESSION_ID_PLACEHOLDER) + .join(sessionEnvironment?.sessionId ?? "") + .split(CWD_PLACEHOLDER) + .join(sessionEnvironment?.cwd ?? "") + .split(PLATFORM_PLACEHOLDER) + .join(sessionEnvironment?.platform ?? "") + .split(OS_VERSION_PLACEHOLDER) + .join(sessionEnvironment?.osVersion ?? "") + .split(DATE_PLACEHOLDER) + .join(sessionEnvironment?.date ?? "") + .trim(); +} + +/** + * Builds the `ToolConfig` needed by Environment from Agent State. + * + * Both builtin tools and MCP Server config are taken from `system_config.yaml`; falls back to the + * default config when builtin tools are missing. + */ +/** + * Filters builtin tool entries by the session model's type: entries with `forModel: "vision"` are + * only used for models that support images (vision models), `forModel: "text-only"` is only for + * text-only models (e.g. choosing between read_image / describe_image); unlabeled entries are + * available to all models. + * Docs: /docs/tools § "Image tools". + */ +export function selectBuiltinToolsForModel( + tools: ToolDefinitionConfig[], + modelVision: boolean, +): ToolDefinitionConfig[] { + const kind = modelVision ? "vision" : "text-only"; + return tools.filter((t) => t.forModel === undefined || t.forModel === kind); +} + +export function buildToolConfig(state: AgentState): ToolConfig { + const systemTools = state.systemConfig.tools; + const builtin = systemTools?.builtin ?? defaultSystemConfig().tools?.builtin ?? []; + return { + customTools: builtin, + mcpServers: systemTools?.mcpServers ?? [], + }; +} + +async function fileExists(filePath: string): Promise { + try { + await fs.access(filePath); + return true; + } catch { + return false; + } +} diff --git a/packages/core/src/state/agent-vault.ts b/packages/core/src/state/agent-vault.ts new file mode 100644 index 0000000..9292a3f --- /dev/null +++ b/packages/core/src/state/agent-vault.ts @@ -0,0 +1,146 @@ +/** + * Agent-level environment-variable vault (`/agents//agent_state/.vault.toml`). + * + * Key-value pairs such as third-party API keys, configured per Agent: injected into that Agent + * session's `exec_command` / `input_command` child-process environment, with key names disclosed + * to the model via the system Prompt while values never enter the model context. Carries the same + * trade-offs as a credential: stored in plaintext on disk, masked at the API layer. The file is + * created/removed together with the Agent directory; its absence is treated as an empty table; + * once emptied, the file is removed to avoid leaving a stray empty .vault.toml. + * Docs: /docs/configuration § "Vault". + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { parse as parseToml, stringify as stringifyToml } from "smol-toml"; +import { agentVaultPath } from "./paths.js"; +import { assertValidId } from "./agent-state.js"; + +/** Vault key-name constraint: matches shell environment variable names (starts with a letter or underscore, followed by letters/digits/underscores only). */ +const VAULT_KEY_PATTERN = /^[A-Za-z_][A-Za-z0-9_]*$/; + +/** + * Vault value length cap: since values are injected into the child-process environment, Linux + * caps a single env entry at roughly 128KB, and an oversized value would make every + * exec_command spawn for that Agent fail (E2BIG) — so it's rejected on the write side (core and + * the API layer share this same cap). + */ +export const VAULT_VALUE_MAX_LENGTH = 8192; + +/** Whether a vault key name is valid (shell environment variable name rules). */ +export function isValidVaultKey(key: string): boolean { + return VAULT_KEY_PATTERN.test(key); +} + +/** Validates a vault key name, throwing if invalid (core and the API layer share this same rule). */ +export function assertValidVaultKey(key: string): void { + if (!isValidVaultKey(key)) { + throw new Error( + `Invalid vault key ${JSON.stringify(key)}: only letters, digits and "_" are allowed, and it must not start with a digit.`, + ); + } +} + +/** Validates a vault value's length (see `VAULT_VALUE_MAX_LENGTH`), throwing if it exceeds the cap. */ +export function assertValidVaultValue(key: string, value: string): void { + if (value.length > VAULT_VALUE_MAX_LENGTH) { + throw new Error( + `Vault value for ${key} is too long: ${value.length} > ${VAULT_VALUE_MAX_LENGTH} characters.`, + ); + } +} + +/** + * Reads the Agent vault: returns an empty table if the file doesn't exist. + * A hand-edited file is filtered by the same rule as the write side: only string values are + * accepted (numbers/dates etc. are ignored), and key names must follow shell variable name rules + * (invalid keys are always ignored — otherwise they'd get injected into the Prompt/child-process + * environment, and an invalid key surfaced by a GET view would make a full-table PUT 400, leaving + * the vault page unable to add or remove any further entries). + */ +export async function loadAgentVault( + root: string, + projectId: string, + agentId: string, +): Promise> { + assertValidId("project_id", projectId); + assertValidId("agent_id", agentId); + let raw: string; + try { + raw = await fs.readFile(agentVaultPath(root, projectId, agentId), "utf8"); + } catch { + return {}; + } + const parsed: unknown = parseToml(raw) ?? {}; + const vault: Record = {}; + if (parsed !== null && typeof parsed === "object" && !Array.isArray(parsed)) { + for (const [k, v] of Object.entries(parsed)) { + if (typeof v === "string" && isValidVaultKey(k)) vault[k] = v; + } + } + return vault; +} + +/** + * Writes the full table to the Agent vault: validates all key names first; an empty table + * deletes the file (idempotent if it doesn't exist). + * The directory is created automatically if it doesn't exist (the vault can be configured even + * before the Agent is initialized). + */ +export async function saveAgentVault( + root: string, + projectId: string, + agentId: string, + vault: Record, +): Promise { + assertValidId("project_id", projectId); + assertValidId("agent_id", agentId); + for (const key of Object.keys(vault)) assertValidVaultKey(key); + const file = agentVaultPath(root, projectId, agentId); + if (Object.keys(vault).length === 0) { + await fs.rm(file, { force: true }); + return; + } + await fs.mkdir(path.dirname(file), { recursive: true }); + // The secret file is written to disk with mode 0600 (a hidden file blocks `ls`, not reads; mode + // only takes effect on creation, so chmod is applied to converge an existing file too). + await fs.writeFile(file, `${stringifyToml(vault)}\n`, { encoding: "utf8", mode: 0o600 }); + await fs.chmod(file, 0o600); +} + +/** + * Writes or updates one vault entry (added if it doesn't exist, overwritten if it does). + * The key name must follow shell environment variable name rules (see `isValidVaultKey`), and the + * value length is constrained by `VAULT_VALUE_MAX_LENGTH`; throws if invalid. Returns the updated + * vault. + */ +export async function setVaultEntry( + root: string, + projectId: string, + agentId: string, + key: string, + value: string, +): Promise> { + assertValidVaultKey(key); + assertValidVaultValue(key, value); + const vault = await loadAgentVault(root, projectId, agentId); + vault[key] = value; + await saveAgentVault(root, projectId, agentId, vault); + return vault; +} + +/** + * Removes one vault entry; idempotent if the key doesn't exist (no write happens). Once emptied, + * the whole .vault.toml is removed. Returns the updated vault. + */ +export async function removeVaultEntry( + root: string, + projectId: string, + agentId: string, + key: string, +): Promise> { + const vault = await loadAgentVault(root, projectId, agentId); + if (!(key in vault)) return vault; + delete vault[key]; + await saveAgentVault(root, projectId, agentId, vault); + return vault; +} diff --git a/packages/core/src/state/builtin-agents.ts b/packages/core/src/state/builtin-agents.ts new file mode 100644 index 0000000..bf76a41 --- /dev/null +++ b/packages/core/src/state/builtin-agents.ts @@ -0,0 +1,48 @@ +/** + * Preset content for builtin Agents; Skill documentation lives in @prismshadow/penguin-skills + * (the library files are read live when building the preset). + * + * - Every Project comes with a single builtin Agent: `default_agent` (the General Agent, the + * default conversational Agent), which has every Skill in the library installed at + * initialization. Dedicated capabilities (creating an Agent, optimizing an Agent, etc.) + * are carried by Skills rather than dedicated builtin Agents. + * - The preset carries no AGENTS.md: the default AGENTS.md is empty, with delegation and task + * conventions living in the default template's Suggested Workflows section. + * - Skill metadata is auto-injected into the system Prompt via the `{{SKILL_METADATA}}` + * placeholder; it's not registered in AGENTS.md. + */ +import { loadLibrarySkills, type LibrarySkill } from "@prismshadow/penguin-skills"; +import { DEFAULT_AGENT_ID } from "./paths.js"; + +/** The set of Project builtin Agent ids (supplied along with the Project, cannot be deleted from Web). */ +export const BUILTIN_AGENT_IDS: readonly string[] = [DEFAULT_AGENT_ID]; + +/** Agent initialization preset (only takes effect at initialization; ignored when loading an existing Agent). */ +export interface AgentPreset { + /** Display name written to system_config.yaml. */ + name?: string; + /** Description written to system_config.yaml. */ + description?: string; + /** Overrides the default AGENTS.md content. */ + agentsMd?: string; + /** Skills installed at initialization (installs none by default). */ + skills?: LibrarySkill[]; +} + +/** + * The preset list for a Project's builtin Agents (each initialized in turn when the Project is + * created; an existing Agent is never overwritten). The only builtin Agent is default_agent: + * installs every Skill in the library, with no preset AGENTS.md. + */ +export function builtinProjectAgentPresets(): Array<{ agentId: string; preset: AgentPreset }> { + return [ + { + agentId: DEFAULT_AGENT_ID, + preset: { + name: "General Agent", + description: "General-purpose agent that completes the user's requests with its tools.", + skills: loadLibrarySkills(), + }, + }, + ]; +} diff --git a/packages/core/src/state/default-config.ts b/packages/core/src/state/default-config.ts new file mode 100644 index 0000000..62a73d8 --- /dev/null +++ b/packages/core/src/state/default-config.ts @@ -0,0 +1,393 @@ +/** + * Default system configuration for Agent State (written to `system_config.yaml`) and the + * default `AGENTS.md` (empty). + * + * Runtime Prompt and tool configuration should come from editable files; + * code only supplies the initial defaults. `system_config.yaml` holds the relatively stable + * system-level Prompt, built-in tools, and MCP Server configuration; `AGENTS.md` is injected + * via a system Prompt placeholder. + * + * The system Prompt is sectioned and trimmed as needed (Role/Personality/Success + * criteria/Constraints/Stop rules/File system/Suggested workflows); it does not describe + * specific tools (that comes from the tool schema). AGENTS.md, Vault/Skills, and Environment + * injection go at the end. + * + * Placeholders (`{{...}}`) appear only in the trailing injection zones (AGENTS.md / Vault / + * Skills / Environment); elsewhere the body uses angle-bracket notation such as + * \`\`, \`\`, \`\` — these are **not substituted**; the model + * fills in the actual values from the Environment section itself. + */ +import type { MCPServerConfig, ThinkingLevelName, ToolDefinitionConfig } from "../interfaces.js"; +import type { CompactionMode } from "../omnimessage/types.js"; + +/** Docs: /docs/configuration § "System prompt placeholders". */ +export const AGENTS_MD_PLACEHOLDER = "{{AGENTS_MD}}"; +export const VAULT_KEYS_PLACEHOLDER = "{{VAULT_KEYS}}"; +export const SKILL_METADATA_PLACEHOLDER = "{{SKILL_METADATA}}"; +export const SESSION_ID_PLACEHOLDER = "{{SESSION_ID}}"; +export const CWD_PLACEHOLDER = "{{CWD}}"; +export const AGENT_ID_PLACEHOLDER = "{{AGENT_ID}}"; +export const PROJECT_DIR_PLACEHOLDER = "{{PROJECT_DIR}}"; +export const PLATFORM_PLACEHOLDER = "{{PLATFORM}}"; +export const OS_VERSION_PLACEHOLDER = "{{OS_VERSION}}"; +export const DATE_PLACEHOLDER = "{{DATE}}"; + +/** + * Context compaction config (the `compaction` section of `system_config.yaml`). + * Docs: /docs/configuration § "Agent config". + */ +export interface CompactionConfig { + /** Context Token threshold (taken from the most recent token_usage's request.total); defaults to 128000, <=0 disables. */ + max_context_length?: number; + /** Session cumulative turn threshold (counted in LLM Requests, across Tasks); defaults to -1, <=0 means no limit. */ + max_session_turns?: number; + /** Compaction mode; defaults to summarize. */ + mode?: CompactionMode; + /** Prompt template for summarize compaction; defaults to the built-in value (editable config, not hardcoded). */ + prompt?: string; +} + +/** + * System-level config for Agent State, serialized as `system_config.yaml`. + * Docs: /docs/configuration § "Agent config". + */ +export interface SystemConfig { + /** Agent display name (display name is separate from id; falls back to id when unset). */ + name?: string; + /** Agent description. */ + description?: string; + /** Agent State version number: a natural number, 1 on creation, incremented on successful optimization; a missing field is treated as 1. */ + version?: number; + /** System-level Prompt (relatively stable; should not be modified frequently). */ + system_prompt: string; + /** Max LLM turns per Task (a runtime parameter that belongs to Agent config, not specified when creating a Session). */ + max_turns?: number; + model?: { + max_tokens?: number; + thinking_level?: ThinkingLevelName; + timeoutMs?: number; + }; + /** Context compaction (enabled by default, max_context_length 128k, mode summarize). */ + compaction?: CompactionConfig; + tools?: { + /** Built-in system tool configuration. */ + builtin?: ToolDefinitionConfig[]; + /** MCP Server configuration. */ + mcpServers?: MCPServerConfig[]; + }; +} + +const DEFAULT_SYSTEM_PROMPT = `# Role +You are PenguinHarness, an agent that completes the user's requests on their machine with the tools available to you. + +# Personality +Communicate with the user precisely and concisely, yet with warmth. Do not repeatedly explain your tools or restate their results. + +# Success criteria +- Before delivering the result, check that every problem in the request has been solved. +- Verify your work through every available means; never claim a result you did not observe. + +# Constraints +- Make the smallest change that satisfies the request; do not modify unrelated files. +- Destructive operations are forbidden. +- Never kill a process you did not start yourself (e.g. to free a busy port) unless the user explicitly asks you to. +- If a tool call fails, read the error, adjust, and retry; never repeat the same failing input. + +# Stop rules +- Stop and give the final answer once the success criteria are met. +- If the request is ambiguous, stop and ask the user for clarification instead of guessing their intent. +- If you hit an error you cannot resolve, stop and report the blocker to the user. + +# Tool use +- Prefer solving problems with your tools: inspect the real files and environment and run real commands instead of answering from memory or guessing. +- When you need information from the internet, browse it with your shell tool — \`curl\` for pages and APIs, or Playwright (if installed) for dynamic sites. + +# System markers +Some user-side messages are system-synthesized records, not user text to answer directly: +- \`\`: the previous round was interrupted. Inside are the original request, your partial thinking/text, and the tool calls already issued with their results. Continue from where it left off; do not re-run tools whose results are already included. +- \`\`: the previous attempt of this round failed on a transport error (timeout or malformed response) — the user did NOT interrupt — and this request is the automatic retry. Inside are your partial thinking/text and the tool calls already executed with their results. Continue from them; do not re-run tools whose results are already included. +- \`\`: earlier conversation was compacted. This summary replaced the raw transcript and is its only record; treat it as established context and continue the task from it. + +# File system +- Angle-bracket markers such as \`\`, \`\` and \`\` are not literal paths — substitute the matching values from the Environment section. +- You run inside the user's working folder (\`CWD\` in Environment). +- The project directory is \`\`; every agent of this project lives under \`/agents/\`, so another agent's assets are at \`/agents//agent_state/\`. +- Your own Agent State is \`/agents//agent_state/\` — it holds your assets such as \`skills/\`, and its \`AGENTS.md\` is already included in your context. Reach these paths directly. +- For temporary and scratch files, create a subdirectory named after the current Session ID under your scratchpad: \`/agents//scratchpad//\`. Build intermediates there, but always place final deliverables in the workspace (under \`CWD\`) — files left in the scratchpad are not part of your output. +- When you create or update a file in the workspace, mention its workspace-relative path in backticks (e.g. \`src/app.py\`) in your reply, so the user can open it from the message. +- Never read, copy, print or otherwise access \`/.project_config.toml\` or any agent's \`agent_state/.vault.toml\` — they hold the user's API keys and other secrets, which are none of your business. Configuration is CLI-only: change models or credentials with \`penguin config ...\` commands. If a task seems to require these files, say so and ask the user instead. + +# Suggested workflows +These are recommendations, not requirements; adapt them as the task demands. +- For a long-horizon task, first write a plan in Markdown to \`/agents//scratchpad//PLAN.md\`, containing a task overview and an itemized step-by-step plan; update it after each completed step to keep execution consistent. +- Delegate self-contained subtasks to other agents with the \`run_subagent\` tool; dispatch independent subtasks in parallel. Start every delegation prompt with your own agent id (e.g. "Caller agent: ") and name the skill the subagent should use when the task matches one. Subagents share your Workspace — exchange data through files. If \`run_subagent\` is not in your tool list, you are the subagent: do the work yourself. +- To visit web pages, prefer Playwright when installed; otherwise \`curl\`. When building a web app or frontend, prefer React. + + +Custom instructions from the developer-editable AGENTS.md. + +{{AGENTS_MD}} + + +# Vault +The vault holds this agent's per-agent secrets (agent_state/.vault.toml). Each entry is injected into your shell subprocesses as an environment variable — values never appear in your context. Use the variable names below in commands when a task needs them. +{{VAULT_KEYS}} + +# Skills +Skills are reusable instruction packages stored under /agents//agent_state/skills//SKILL.md. There is no skill tool: when a task matches an installed skill below, or the user asks to use one (a message may start with a block listing skill names), first read that skill's SKILL.md in full with a shell command, then follow it. If a request only names a skill without a concrete task, ask the user what they need before starting. +{{SKILL_METADATA}} + +# Environment +- Platform: {{PLATFORM}} +- OS Version: {{OS_VERSION}} +- Date: {{DATE}} +- CWD: {{CWD}} +- Agent ID: {{AGENT_ID}} +- Project Dir: {{PROJECT_DIR}} +- Session ID: {{SESSION_ID}}`; + +/** + * Built-in default compaction Prompt (summarize mode): tells the model that after + * compaction the raw transcript is no longer visible and the + * summary is the only record, so it must include everything needed to continue the task, + * and no tools may be called while writing the summary. + */ +export const DEFAULT_COMPACTION_PROMPT = + "You have a partial transcript of the task above. Write a summary of it wrapped in " + + "`

` tags. This summary will replace the transcript: in the next " + + "context window the raw transcript above will no longer be visible and this summary " + + "will be its only record, so include everything needed to continue the task — the " + + "original request, current state, next steps, and any learnings. Do not call any " + + "tools while writing the summary; respond with text only."; + +/** + * Default built-in system tools: bash execution and subagent spawning. + * Docs: /docs/tools § "Built-in tools". + */ +function defaultBuiltinTools(): ToolDefinitionConfig[] { + return [ + { + name: "exec_command", + description: + "Run a shell command in the workspace to read, write, edit files and run programs. " + + "Run long-lived commands (servers, watchers, builds) in the foreground: past yield_time_ms " + + "they keep running in the background with a process_id. Do not background them with `&` — " + + "the whole process group is cleaned up when the foreground command exits.", + parameters: { + type: "object", + properties: { + cmd: { + type: "string", + description: "Shell command to execute.", + }, + workdir: { + type: "string", + description: + "Working directory for the command; defaults to the cwd. Optionally a path relative to the cwd, or an absolute path.", + }, + yield_time_ms: { + type: "number", + description: + "How long to wait for the command before yielding. If it is still running when this elapses, the tool returns the output so far plus a process_id, and the command keeps running in the background (drive it with input_command). Defaults to 60000; minimum 250, capped below the tool timeout.", + }, + }, + required: ["cmd"], + }, + permission: "rw", + timeoutMs: 120000, + maxOutputLength: 16000, + }, + { + name: "input_command", + description: + "Interact with a running command session started by exec_command: write to its stdin, send Ctrl-C, or poll for new output. Identify the session with its process_id.", + parameters: { + type: "object", + properties: { + process_id: { + type: "string", + description: "The process_id returned by exec_command for the running command session.", + }, + chars: { + type: "string", + description: + 'Characters to write to the command\'s stdin. Send "\\u0003" alone to deliver Ctrl-C (SIGINT); mixing it with other characters is an error. Empty (the default) writes nothing and only polls for new output and exit status.', + }, + yield_time_ms: { + type: "number", + description: + "How long to wait for new output or exit before returning. Non-empty writes default to 250; empty polls default to 5000. Minimum 250, capped below the tool timeout.", + }, + }, + required: ["process_id"], + }, + permission: "rw", + // An empty poll can wait out a build/test run (the yield ceiling is derived from timeoutMs, clamped inside the tool). + timeoutMs: 130000, + maxOutputLength: 16000, + }, + { + name: "run_subagent", + description: + "Delegate a self-contained subtask to a subagent that runs autonomously in the same workspace and returns its final answer. Use it for focused sub-tasks you can fully specify in one prompt. Optionally choose a specific agent via `agent_id` and a model via `model_id`. " + + 'Begin the prompt by identifying yourself with your own agent id (from the Environment section), e.g. "Caller agent: default_agent" — the subagent cannot otherwise tell who invoked it.', + parameters: { + type: "object", + properties: { + prompt: { + type: "string", + description: + "The complete task for the subagent: include all context it needs and the exact final output you expect back.", + }, + agent_id: { + type: "string", + description: + "Which agent to run as the subagent; defaults to the current agent when omitted.", + }, + model_id: { + type: "string", + description: + "Which model the subagent should use; defaults to the Project default model when omitted.", + }, + yield_time_ms: { + type: "number", + description: + "How long to wait for the subagent before yielding. If it is still working when this elapses, the tool returns the output so far plus a subagent_id, and the subagent keeps running in the background (drive it with input_subagent). Defaults to 300000; minimum 250, capped below the tool timeout.", + }, + }, + required: ["prompt"], + }, + permission: "rw", + // Subagent tasks typically run far longer than a single command, so the timeout ceiling is raised accordingly. + timeoutMs: 600000, + maxOutputLength: 16000, + }, + { + name: "input_subagent", + description: + "Interact with a background subagent started by run_subagent: poll for new output, or send a follow-up prompt once it is idle to continue the same subagent session. Identify the session with its subagent_id. Pending tool approvals of the subagent are surfaced while this tool is waiting.", + parameters: { + type: "object", + properties: { + subagent_id: { + type: "string", + description: "The subagent_id returned by run_subagent for the background subagent.", + }, + prompt: { + type: "string", + description: + "A follow-up task for the subagent, delivered as a new user message on the same session. Only accepted when the subagent is idle (its previous run finished). Empty (the default) sends nothing and only polls for new output and status.", + }, + yield_time_ms: { + type: "number", + description: + "How long to wait for new output or completion before returning. Follow-up prompts default to 300000; empty polls default to 10000. Minimum 250, capped below the tool timeout.", + }, + }, + required: ["subagent_id"], + }, + permission: "rw", + // Same generous timeout tier as run_subagent: an empty poll can wait a long time for the subagent to wrap up. + timeoutMs: 600000, + maxOutputLength: 16000, + }, + // The image-reading tools are mutually exclusive based on the session model's type + // (marked via each entry's forModel, filtered at assembly time): read_image is designed + // for vision models (the image is fed back as image content); describe_image is designed + // for text-only models (the image plus the prompt are sent to the Project's configured + // vision model, vision_model, whose text answer becomes the tool output). + { + name: "read_image", + forModel: "vision", + description: + "Read an image and return it as image content for you to view. Accepts an http(s) URL " + + "or a local file path (relative paths resolve against the workspace). " + + "Supports png/jpeg/gif/webp up to 5MB.", + parameters: { + type: "object", + properties: { + source: { + type: "string", + description: + "Image to read: an http(s) URL, or a local file path (absolute, or relative to the workspace).", + }, + }, + required: ["source"], + }, + permission: "r", + timeoutMs: 60000, + maxOutputLength: 16000, + }, + { + name: "describe_image", + forModel: "text-only", + description: + "Describe an image and return a TEXT description of it. The current model does not accept " + + "images directly, so the image is analyzed by the project's configured vision model and " + + "you get its text answer back. Use `prompt` to ask exactly what you need to know about " + + "the image (e.g. transcribe text, describe a chart, locate a UI element). Accepts an " + + "http(s) URL or a local file path (relative paths resolve against the workspace). " + + "Supports png/jpeg/gif/webp up to 5MB.", + parameters: { + type: "object", + properties: { + source: { + type: "string", + description: + "Image to read: an http(s) URL, or a local file path (absolute, or relative to the workspace).", + }, + prompt: { + type: "string", + description: + "What to ask about the image; the vision model answers this. Defaults to a detailed description.", + }, + }, + required: ["source"], + }, + permission: "r", + // Includes one vision-model request, so the timeout is slightly wider than plain image reading. + timeoutMs: 90000, + maxOutputLength: 16000, + }, + ]; +} + +/** Agent State version number: an invalid or missing field is always treated as 1. */ +export function agentStateVersion(config: Pick): number { + const v = config.version; + return typeof v === "number" && Number.isInteger(v) && v >= 1 ? v : 1; +} + +/** Returns the default system configuration for Agent State. */ +export function defaultSystemConfig(): SystemConfig { + return { + version: 1, + system_prompt: DEFAULT_SYSTEM_PROMPT, + max_turns: 100, + model: { + max_tokens: 32000, + thinking_level: "medium", + timeoutMs: 120000, + }, + compaction: { + max_context_length: 128000, + max_session_turns: -1, + mode: "summarize", + prompt: DEFAULT_COMPACTION_PROMPT, + }, + tools: { + builtin: defaultBuiltinTools(), + mcpServers: [], + }, + }; +} + +/** + * Returns the default editable `AGENTS.md` content: an empty string — no guidance is + * preprovisioned by default; Subagent delegation conventions and general task practices + * live in the default template's Suggested workflows section as a soft convention. + * Kept so initialization can still write an empty AGENTS.md file. + */ +export function defaultAgentsMd(): string { + return ""; +} diff --git a/packages/core/src/state/example-benchmark.ts b/packages/core/src/state/example-benchmark.ts new file mode 100644 index 0000000..2e53795 --- /dev/null +++ b/packages/core/src/state/example-benchmark.ts @@ -0,0 +1,345 @@ +/** + * Provisioning of the example Benchmark. + * + * When default_agent is initialized, it preprovisions `benchmarks/example-benchmark/`: two + * sample cases (each with statement/ and rubric/ indexed by a README.md), + * benchmark_config.toml (runs = 2), and a scoreboard.yaml with three sample evaluations — + * so the evaluation center has data out of the box. Its description states plainly that this + * is a built-in example and the whole directory can be deleted or replaced. Only + * default_agent gets this; ordinary Agents do not. + * + * Scoring numbers are self-consistent: each case's score / cost / duration_ms is the + * **average** computed from its runs array, and each evaluation's totals are the sum over + * its cases (written this way so it already satisfies the scoreboard v2 convention, and + * tests can verify it). + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { stringify as stringifyToml } from "smol-toml"; +import { stringify as stringifyYaml } from "yaml"; +import { benchmarksDir } from "./paths.js"; + +/** Directory name of the example Benchmark (the directory name is also its identifier). */ +export const EXAMPLE_BENCHMARK_ID = "example-benchmark"; + +/** Contents of benchmark_config.toml (no model reference here — the model is recorded on each evaluation instead). */ +const EXAMPLE_BENCHMARK_CONFIG = { + title: "Example Benchmark", + description: + "A built-in example benchmark so the evaluation charts have data out of the box. " + + "Replace it with your own.", + runs: 2, +}; + +/** Two sample cases: statement and scoring rubric (in English, 3-5 lines each). */ +const EXAMPLE_CASES: Array<{ id: string; statement: string; rubric: string }> = [ + { + id: "CASE-001-file-summary", + statement: `# Task: Summarize a project file + +Read the provided \`notes.txt\` in your workspace and write \`summary.md\` containing: +1. A one-paragraph overview of at most 3 sentences. +2. A bullet list of the three most important facts. +Keep the whole summary under 150 words. +`, + rubric: `# Scoring rubric (max 5 points) + +- 2 pts: \`summary.md\` exists and stays under 150 words. +- 2 pts: The three bullet facts are accurate and taken from \`notes.txt\`. +- 1 pt: The overview paragraph is coherent and at most 3 sentences. +Award partial credit per item; the case score is the sum. +`, + }, + { + id: "CASE-002-data-cleanup", + statement: `# Task: Clean up a CSV dataset + +The workspace contains \`users.csv\` with duplicate rows and inconsistent casing in the email column. +Produce \`users_clean.csv\` where: +1. Emails are lowercased and rows with an empty email are removed. +2. Exact duplicate rows are dropped, keeping the first occurrence. +Do not change the column order. +`, + rubric: `# Scoring rubric (max 5 points) + +- 2 pts: \`users_clean.csv\` exists and keeps the original column order. +- 2 pts: Emails are lowercased, empty-email rows removed, duplicates dropped (first kept). +- 1 pt: No unrelated rows or columns were modified. +Award partial credit per item; the case score is the sum. +`, + }, +]; + +/** Raw result of a single run (a runs element in scoreboard v2). */ +interface ExampleRun { + score: number; + cost: number; + duration_ms: number; + session_id: string; +} + +/** + * Raw runs for the three sample evaluations (case-level and evaluation-level metrics are + * computed from these, keeping the numbers self-consistent). Each carries the model actually + * used for that round (paired, since the evaluation center's chart splits series by model); + * the examples all use deepseek-v4-pro (a single model, single series). + */ +const EXAMPLE_EVALUATIONS: Array<{ + time: string; + version: number; + provider: string; + model_id: string; + summary_title: string; + summary: string; + cases: Array<{ case: string; runs: ExampleRun[] }>; +}> = [ + { + time: "2026-07-14T09:30:00Z", + version: 1, + provider: "deepseek", + model_id: "deepseek-v4-pro", + summary_title: "Baseline before any optimization", + summary: + "Example data (not a real evaluation): baseline scores of the built-in sample " + + "benchmark before any optimization. Hypothesis for the next round: the agent skips " + + "a final self-check, losing points on completeness.", + cases: [ + { + case: "CASE-001-file-summary", + runs: [ + { + score: 2.5, + cost: 0.012, + duration_ms: 42000, + session_id: "session-2026-07-14-09-05-11-1a2b3c01", + }, + { + score: 3.5, + cost: 0.014, + duration_ms: 48000, + session_id: "session-2026-07-14-09-13-27-1a2b3c02", + }, + ], + }, + { + case: "CASE-002-data-cleanup", + runs: [ + { + score: 3.0, + cost: 0.018, + duration_ms: 66000, + session_id: "session-2026-07-14-09-21-45-1a2b3c03", + }, + { + score: 3.0, + cost: 0.022, + duration_ms: 74000, + session_id: "session-2026-07-14-09-28-52-1a2b3c04", + }, + ], + }, + ], + }, + { + time: "2026-07-15T09:30:00Z", + version: 2, + provider: "deepseek", + model_id: "deepseek-v4-pro", + summary_title: "Added an explicit planning step", + summary: + "Example data (not a real evaluation): after adding an explicit planning step to the " + + "system prompt (hypothesis: written plans reduce missed requirements), both cases " + + "improved. Next: tighten output formatting.", + cases: [ + { + case: "CASE-001-file-summary", + runs: [ + { + score: 3.5, + cost: 0.011, + duration_ms: 39000, + session_id: "session-2026-07-15-09-04-33-2b3c4d01", + }, + { + score: 4.0, + cost: 0.013, + duration_ms: 45000, + session_id: "session-2026-07-15-09-12-08-2b3c4d02", + }, + ], + }, + { + case: "CASE-002-data-cleanup", + runs: [ + { + score: 3.5, + cost: 0.016, + duration_ms: 60000, + session_id: "session-2026-07-15-09-19-40-2b3c4d03", + }, + { + score: 4.0, + cost: 0.02, + duration_ms: 68000, + session_id: "session-2026-07-15-09-26-59-2b3c4d04", + }, + ], + }, + ], + }, + { + time: "2026-07-16T09:30:00Z", + version: 3, + provider: "deepseek", + model_id: "deepseek-v4-pro", + summary_title: "Verify deliverables before finishing", + summary: + "Example data (not a real evaluation): after instructing the agent to verify its " + + "deliverables against the statement before finishing (hypothesis: a final check " + + "catches formatting slips), scores improved again. Replace this benchmark with your " + + "own to track real progress.", + cases: [ + { + case: "CASE-001-file-summary", + runs: [ + { + score: 4.0, + cost: 0.01, + duration_ms: 36000, + session_id: "session-2026-07-16-09-03-21-3c4d5e01", + }, + { + score: 4.5, + cost: 0.012, + duration_ms: 40000, + session_id: "session-2026-07-16-09-10-46-3c4d5e02", + }, + ], + }, + { + case: "CASE-002-data-cleanup", + runs: [ + { + score: 4.5, + cost: 0.015, + duration_ms: 55000, + session_id: "session-2026-07-16-09-18-02-3c4d5e03", + }, + { + score: 4.0, + cost: 0.017, + duration_ms: 61000, + session_id: "session-2026-07-16-09-25-30-3c4d5e04", + }, + ], + }, + ], + }, +]; + +/** Round floats to 1e-6 (so binary error from averaging/summing isn't persisted to disk). */ +function round(v: number): number { + return Math.round(v * 1e6) / 1e6; +} + +function average(values: number[]): number { + return round(values.reduce((a, b) => a + b, 0) / values.length); +} + +function sum(values: number[]): number { + return round(values.reduce((a, b) => a + b, 0)); +} + +/** + * Builds the scoreboard object from raw runs data: each case's three metrics are the + * average of its runs, and each evaluation's metrics are the sum of its cases' averages + * (following the scoreboard v2 convention). Exported so tests can verify the numbers + * are self-consistent. + */ +export function buildExampleScoreboard(): { + evaluations: Array<{ + time: string; + version: number; + provider: string; + model_id: string; + summary_title: string; + summary: string; + score: number; + cost: number; + duration_ms: number; + cases: Array<{ + case: string; + score: number; + cost: number; + duration_ms: number; + runs: ExampleRun[]; + }>; + }>; +} { + return { + evaluations: EXAMPLE_EVALUATIONS.map((e) => { + const cases = e.cases.map((c) => ({ + case: c.case, + score: average(c.runs.map((r) => r.score)), + cost: average(c.runs.map((r) => r.cost)), + duration_ms: average(c.runs.map((r) => r.duration_ms)), + runs: c.runs, + })); + return { + time: e.time, + version: e.version, + provider: e.provider, + model_id: e.model_id, + summary_title: e.summary_title, + summary: e.summary, + score: sum(cases.map((c) => c.score)), + cost: sum(cases.map((c) => c.cost)), + duration_ms: sum(cases.map((c) => c.duration_ms)), + cases, + }; + }), + }; +} + +/** + * Provisions the example Benchmark: if `benchmarks/` already exists (the user already has a + * case library), does nothing; otherwise creates `benchmarks/example-benchmark/` (config, the + * two sample cases, and the scoreboard). Callers are restricted to the default_agent + * initialization path (see agent-state.ts). + */ +export async function provisionExampleBenchmark( + root: string, + projectId: string, + agentId: string, +): Promise { + const dir = benchmarksDir(root, projectId, agentId); + try { + await fs.access(dir); + return; + } catch { + // benchmarks/ does not exist: proceed with provisioning. + } + const benchDir = path.join(dir, EXAMPLE_BENCHMARK_ID); + await Promise.all( + EXAMPLE_CASES.flatMap((c) => [ + fs.mkdir(path.join(benchDir, c.id, "statement"), { recursive: true }), + fs.mkdir(path.join(benchDir, c.id, "rubric"), { recursive: true }), + ]), + ); + await Promise.all([ + fs.writeFile( + path.join(benchDir, "benchmark_config.toml"), + `${stringifyToml(EXAMPLE_BENCHMARK_CONFIG)}\n`, + "utf8", + ), + fs.writeFile( + path.join(benchDir, "scoreboard.yaml"), + stringifyYaml(buildExampleScoreboard()), + "utf8", + ), + ...EXAMPLE_CASES.flatMap((c) => [ + fs.writeFile(path.join(benchDir, c.id, "statement", "README.md"), c.statement, "utf8"), + fs.writeFile(path.join(benchDir, c.id, "rubric", "README.md"), c.rubric, "utf8"), + ]), + ]); +} diff --git a/packages/core/src/state/index.ts b/packages/core/src/state/index.ts new file mode 100644 index 0000000..aacd1c3 --- /dev/null +++ b/packages/core/src/state/index.ts @@ -0,0 +1,16 @@ +/** + * Agent State and Project config storage. + * + * Directory layout, default config, Project config read/write, Agent State load/init. + */ +export * from "./paths.js"; +export * from "./default-config.js"; +export * from "./builtin-agents.js"; +export * from "./model-catalog.js"; +export * from "./project-config.js"; +export * from "./agent-state.js"; +export * from "./agent-vault.js"; +export * from "./example-benchmark.js"; + +// Skill library types and frontmatter parser (from the skills package; server reuses the same implementation via core). +export { parseSkillFrontmatter, type SkillMetadata } from "@prismshadow/penguin-skills"; diff --git a/packages/core/src/state/model-catalog.ts b/packages/core/src/state/model-catalog.ts new file mode 100644 index 0000000..fce375c --- /dev/null +++ b/packages/core/src/state/model-catalog.ts @@ -0,0 +1,489 @@ +/** + * Built-in model catalog (single source of truth): official chat models that AgentHub can + * auto-route, shared by core's default config, server's initial config, and web/cli display. + * Data verified as of 2026-07-10. + * Docs: packages/docs/content/models.{zh,en}.md (site path /docs/models) documents the + * provider groups and credential resolution described here. + * + * Three-bucket pricing convention (USD per million tokens, matching usageToTokenCounts' + * token-to-bucket mapping): + * - cache_read: the vendor's "cache hit" price; + * - cache_write: the vendor's "cache write" price (e.g. Anthropic uses 1.25 x input); vendors + * without a separate cache-write fee use the standard input price; + * - output: output price (thinking + reply). + * OpenAI charges extra for >272K input and Gemini 3.1 Pro for >200K input under official + * long-context pricing; this catalog only records the base tier (the cost center uses a + * single rate, so long-context usage will be underestimated). + * + * Scope: excludes deepseek-chat / deepseek-reasoner legacy aliases that AgentHub cannot + * auto-route (deprecated 2026-07-24), glm-5v-turbo (image input unsupported by AgentHub's GLM + * client), non-chat models (embedding / image generation / TTS), and Bedrock plus + * OpenRouter / SiliconFlow gateway mirror ids. Every model id in this catalog can be + * auto-routed by AgentHub via substring matching, so none set client_type; only custom + * OpenAI-protocol models need `client_type: "openai"`. + * + * This file imports no Node built-ins (type-only imports only), so it can be bundled directly + * for the browser. + */ +import type { ModelEntry, ModelPricing } from "./project-config.js"; + +/** Model provider info (used for web grouping/logo and the "API key blank falls back to env var" hint). */ +export interface ModelProviderInfo { + id: string; + /** Display name (brand name, shared by Chinese and English UI). */ + label: string; + /** API key env var name (AgentHub reads this automatically when credential is blank). */ + envKey: string; + /** base URL env var name. */ + envBaseUrlKey: string; + /** Console URL for obtaining an API key (frontend links this in the group header); none for custom. */ + apiKeyUrl?: string; + /** Vendor's model list / docs page URL (frontend's "add model" dialog links this as "get model id"); none for custom. */ + modelsUrl?: string; + /** + * Gateway's OpenAI-compatible endpoint (openrouter / siliconflow): used by the frontend's + * "add model" dialog to prefill base URL by group; left blank for direct vendors and custom. + */ + gatewayBaseUrl?: string; +} + +/** A single built-in model's catalog entry (`modelId` is the upstream id; paired with `provider` it forms the catalog's unique key). */ +export interface ModelCatalogEntry { + modelId: string; + displayName: string; + /** Provider id (one of MODEL_PROVIDERS). */ + provider: string; + contextWindow?: number; + pricing?: ModelPricing; + /** Whether image input (vision modality) is supported. */ + supportsVision: boolean; + /** AgentHub client protocol: required for models whose id can't be auto-routed (e.g. OpenRouter gateway models). */ + clientType?: string; + /** Preset base URL (gateway models): inlined into the model entry so the user only needs to supply an API key. */ + baseUrl?: string; +} + +/** Each gateway's OpenAI-compatible endpoint (preset base URL for gateway models; also used as the provider's gatewayBaseUrl). */ +const OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1"; +const SILICONFLOW_BASE_URL = "https://api.siliconflow.cn/v1"; + +/** + * Provider list (web model page groups in this order): DeepSeek first (the default model's + * provider), followed by the OpenRouter and SiliconFlow gateways, then Google Gemini before + * Anthropic; custom groups custom OpenAI-protocol models and comes last. + */ +export const MODEL_PROVIDERS: ModelProviderInfo[] = [ + { + id: "deepseek", + label: "DeepSeek", + envKey: "DEEPSEEK_API_KEY", + envBaseUrlKey: "DEEPSEEK_BASE_URL", + apiKeyUrl: "https://platform.deepseek.com/api_keys", + modelsUrl: "https://api-docs.deepseek.com/quick_start/pricing", + }, + // Gateways (their model ids can't be auto-routed by AgentHub, so they always use + // client_type=openai + a preset base URL): they go through AgentHub's OpenAI client, so when + // credential is blank the SDK reads **OPENAI_API_KEY / OPENAI_BASE_URL** (not the provider's + // own var names) - the env fallback hint must reflect that accurately. + { + id: "openrouter", + label: "OpenRouter", + envKey: "OPENAI_API_KEY", + envBaseUrlKey: "OPENAI_BASE_URL", + apiKeyUrl: "https://openrouter.ai/workspaces/default/keys", + modelsUrl: "https://openrouter.ai/models", + gatewayBaseUrl: OPENROUTER_BASE_URL, + }, + { + id: "siliconflow", + label: "SiliconFlow", + envKey: "OPENAI_API_KEY", + envBaseUrlKey: "OPENAI_BASE_URL", + apiKeyUrl: "https://cloud.siliconflow.cn/me/account/ak", + modelsUrl: "https://cloud.siliconflow.cn/models", + gatewayBaseUrl: SILICONFLOW_BASE_URL, + }, + { + id: "google", + label: "Google Gemini", + envKey: "GEMINI_API_KEY", + envBaseUrlKey: "GEMINI_BASE_URL", + apiKeyUrl: "https://aistudio.google.com/api-keys", + modelsUrl: "https://ai.google.dev/gemini-api/docs/models", + }, + { + id: "anthropic", + label: "Anthropic", + envKey: "ANTHROPIC_API_KEY", + envBaseUrlKey: "ANTHROPIC_BASE_URL", + apiKeyUrl: "https://platform.claude.com/settings/keys", + modelsUrl: "https://docs.claude.com/en/docs/about-claude/models/overview", + }, + { + id: "openai", + label: "OpenAI", + envKey: "OPENAI_API_KEY", + envBaseUrlKey: "OPENAI_BASE_URL", + apiKeyUrl: "https://platform.openai.com/api-keys", + modelsUrl: "https://platform.openai.com/docs/models", + }, + { + id: "zhipu", + label: "Z.AI (GLM)", + envKey: "ZAI_API_KEY", + envBaseUrlKey: "ZAI_BASE_URL", + apiKeyUrl: "https://open.bigmodel.cn/apikey/platform", + modelsUrl: "https://docs.z.ai/guides/overview/pricing", + }, + { + id: "moonshot", + label: "Moonshot (Kimi)", + envKey: "MOONSHOT_API_KEY", + envBaseUrlKey: "MOONSHOT_BASE_URL", + apiKeyUrl: "https://platform.kimi.com/console/api-keys", + modelsUrl: "https://platform.kimi.com/docs/pricing", + }, + { id: "custom", label: "Custom", envKey: "OPENAI_API_KEY", envBaseUrlKey: "OPENAI_BASE_URL" }, +]; + +/** Three-bucket price literal (unit fixed to usd_per_mtok). */ +/** + * Converts official CNY pricing to USD for storage (prices are always persisted in USD). The + * conversion rate matches the web display's 7:1 convention, so switching the UI to CNY shows + * exactly the vendor's official CNY price. + */ +function cny(cacheRead: number, cacheWrite: number, output: number): ModelPricing { + const r = (v: number): number => Math.round((v / 7) * 1e6) / 1e6; + return usd(r(cacheRead), r(cacheWrite), r(output)); +} + +function usd(cacheRead: number, cacheWrite: number, output: number): ModelPricing { + return { unit: "usd_per_mtok", cache_read: cacheRead, cache_write: cacheWrite, output }; +} + +/** Built-in model catalog (clustered by provider; within each provider, ordered by capability/price, highest first). */ +export const MODEL_CATALOG: ModelCatalogEntry[] = [ + // -- DeepSeek (official CNY pricing: cache hit / cache miss / output) -- + { + modelId: "deepseek-v4-pro", + displayName: "DeepSeek V4 Pro", + provider: "deepseek", + contextWindow: 1000000, + pricing: cny(0.025, 3, 6), + supportsVision: false, + }, + { + modelId: "deepseek-v4-flash", + displayName: "DeepSeek V4 Flash", + provider: "deepseek", + contextWindow: 1000000, + pricing: cny(0.02, 1, 2), + supportsVision: false, + }, + // —— Anthropic —— + { + modelId: "claude-opus-4-8", + displayName: "Claude Opus 4.8", + provider: "anthropic", + contextWindow: 1000000, + pricing: usd(0.5, 6.25, 25), + supportsVision: true, + }, + { + modelId: "claude-opus-4-7", + displayName: "Claude Opus 4.7", + provider: "anthropic", + contextWindow: 1000000, + pricing: usd(0.5, 6.25, 25), + supportsVision: true, + }, + { + modelId: "claude-sonnet-4-6", + displayName: "Claude Sonnet 4.6", + provider: "anthropic", + contextWindow: 1000000, + pricing: usd(0.3, 3.75, 15), + supportsVision: true, + }, + // —— OpenAI —— + { + modelId: "gpt-5.5", + displayName: "GPT-5.5", + provider: "openai", + contextWindow: 1050000, + pricing: usd(0.5, 5, 30), + supportsVision: true, + }, + { + // No official cache discount: cache_read uses the standard input price. + modelId: "gpt-5.5-pro", + displayName: "GPT-5.5 Pro", + provider: "openai", + contextWindow: 1050000, + pricing: usd(30, 30, 180), + supportsVision: true, + }, + { + modelId: "gpt-5.4", + displayName: "GPT-5.4", + provider: "openai", + contextWindow: 1050000, + pricing: usd(0.25, 2.5, 15), + supportsVision: true, + }, + { + modelId: "gpt-5.4-mini", + displayName: "GPT-5.4 mini", + provider: "openai", + contextWindow: 400000, + pricing: usd(0.075, 0.75, 4.5), + supportsVision: true, + }, + { + modelId: "gpt-5.4-nano", + displayName: "GPT-5.4 nano", + provider: "openai", + contextWindow: 400000, + pricing: usd(0.02, 0.2, 1.25), + supportsVision: true, + }, + { + // No official cache discount: cache_read uses the standard input price. + modelId: "gpt-5.4-pro", + displayName: "GPT-5.4 Pro", + provider: "openai", + contextWindow: 1050000, + pricing: usd(30, 30, 180), + supportsVision: true, + }, + // —— Google Gemini —— + { + // ≤200K input tier; >200K has official surcharge pricing (see file header comment). + modelId: "gemini-3.1-pro-preview", + displayName: "Gemini 3.1 Pro (Preview)", + provider: "google", + contextWindow: 1048576, + pricing: usd(0.2, 2, 12), + supportsVision: true, + }, + { + modelId: "gemini-3.5-flash", + displayName: "Gemini 3.5 Flash", + provider: "google", + contextWindow: 1048576, + pricing: usd(0.15, 1.5, 9), + supportsVision: true, + }, + { + modelId: "gemini-3-flash-preview", + displayName: "Gemini 3 Flash (Preview)", + provider: "google", + contextWindow: 1048576, + pricing: usd(0.05, 0.5, 3), + supportsVision: true, + }, + { + modelId: "gemini-3.1-flash-lite", + displayName: "Gemini 3.1 Flash-Lite", + provider: "google", + contextWindow: 1048576, + pricing: usd(0.025, 0.25, 1.5), + supportsVision: true, + }, + // —— Z.AI (GLM) —— + { + modelId: "glm-5.2", + displayName: "GLM-5.2", + provider: "zhipu", + contextWindow: 1000000, + pricing: usd(0.26, 1.4, 4.4), + supportsVision: false, + }, + { + modelId: "glm-5.1", + displayName: "GLM-5.1", + provider: "zhipu", + contextWindow: 200000, + pricing: usd(0.26, 1.4, 4.4), + supportsVision: false, + }, + { + modelId: "glm-5", + displayName: "GLM-5", + provider: "zhipu", + contextWindow: 200000, + pricing: usd(0.2, 1, 3.2), + supportsVision: false, + }, + // -- Moonshot (Kimi) (official CNY pricing) -- + { + modelId: "kimi-k2.6", + displayName: "Kimi K2.6", + provider: "moonshot", + contextWindow: 262144, + pricing: cny(1.1, 6.5, 27), + supportsVision: true, + }, + { + modelId: "kimi-k2.5", + displayName: "Kimi K2.5", + provider: "moonshot", + contextWindow: 262144, + pricing: cny(0.7, 4, 21), + supportsVision: true, + }, + // -- OpenRouter (gateway: uses OpenAI-compatible protocol, preset base URL) -- + { + modelId: "xiaomi/mimo-v2.5", + displayName: "MiMo-V2.5", + provider: "openrouter", + contextWindow: 1048576, + pricing: usd(0.0028, 0.14, 0.28), + supportsVision: true, + clientType: "openai", + baseUrl: OPENROUTER_BASE_URL, + }, + { + modelId: "tencent/hy3", + displayName: "Hy3", + provider: "openrouter", + contextWindow: 262144, + pricing: usd(0.035, 0.14, 0.58), + supportsVision: false, + clientType: "openai", + baseUrl: OPENROUTER_BASE_URL, + }, + { + // No official separate cache price published: cache_read uses the standard input price (no discount assumed). + modelId: "minimax/minimax-m3", + displayName: "MiniMax M3", + provider: "openrouter", + contextWindow: 1048576, + pricing: usd(0.06, 0.3, 1.2), + supportsVision: true, + clientType: "openai", + baseUrl: OPENROUTER_BASE_URL, + }, + { + // No official separate cache price published: cache_read uses the standard input price. + modelId: "stepfun/step-3.7-flash", + displayName: "Step 3.7 Flash", + provider: "openrouter", + contextWindow: 256000, + pricing: usd(0.04, 0.2, 1.15), + supportsVision: true, + clientType: "openai", + baseUrl: OPENROUTER_BASE_URL, + }, + // -- SiliconFlow (gateway, official CNY pricing: cache hit / input / output) -- + { + modelId: "zai-org/GLM-5.2", + displayName: "GLM-5.2", + provider: "siliconflow", + contextWindow: 1000000, + pricing: cny(2, 8, 28), + supportsVision: false, + clientType: "openai", + baseUrl: SILICONFLOW_BASE_URL, + }, + { + modelId: "deepseek-ai/DeepSeek-V4-Pro", + displayName: "DeepSeek V4 Pro", + provider: "siliconflow", + contextWindow: 1000000, + pricing: cny(0.1, 12, 24), + supportsVision: false, + clientType: "openai", + baseUrl: SILICONFLOW_BASE_URL, + }, + { + modelId: "meituan-longcat/LongCat-2.0", + displayName: "LongCat 2.0", + provider: "siliconflow", + contextWindow: 1000000, + pricing: cny(0.1, 5, 20), + supportsVision: false, + clientType: "openai", + baseUrl: SILICONFLOW_BASE_URL, + }, +]; + +/** Looks up a catalog entry by (provider, upstream id) pair (**the sole catalog-matching entry point**); returns undefined if not in the catalog. */ +export function catalogEntryFor( + provider: string, + upstreamId: string, +): ModelCatalogEntry | undefined { + return MODEL_CATALOG.find((m) => m.provider === provider && m.modelId === upstreamId); +} + +/** + * Infers the provider for an upstream id from the built-in catalog (used to default + * `provider` on `model add`): if it matches a catalog entry, use that entry's provider + * (upstream ids are globally unique within the catalog); otherwise custom. + */ +export function inferProviderForUpstream(upstreamId: string): string { + return MODEL_CATALOG.find((m) => m.modelId === upstreamId)?.provider ?? "custom"; +} + +/** Looks up provider info by provider id; returns undefined for an unknown id. */ +export function providerInfo(providerId: string): ModelProviderInfo | undefined { + return MODEL_PROVIDERS.find((p) => p.id === providerId); +} + +/** Env var fallback for a single model (the var names AgentHub's client actually reads when api_key / base_url is blank). */ +export interface ModelEnvInfo { + envKey: string; + envBaseUrlKey: string; +} + +/** + * Resolves the env var fallback for a model: mirrors AgentHub's + * AutoLLMClient routing rules (verified against agenthub v0.3.3 autoClient.ts) - an explicit + * client_type takes priority, otherwise routes to a client by lowercase substring match on + * model_id, returning the var pair that client reads; branch order matches AutoLLMClient. + * Returns undefined on no match (AgentHub will reject that id: it needs an explicit + * client_type, or should be added under custom / a self-built group via the OpenAI protocol). + */ +export function resolveModelEnv(modelId: string, clientType?: string): ModelEnvInfo | undefined { + const t = (clientType || modelId).toLowerCase(); + const env = (prefix: string): ModelEnvInfo => ({ + envKey: `${prefix}_API_KEY`, + envBaseUrlKey: `${prefix}_BASE_URL`, + }); + if (t.includes("gemini-3") || t.includes("gemini-embedding")) return env("GEMINI"); + if ( + t.includes("claude") && + (t.includes("4-7") || t.includes("4-8") || t.includes("-5") || t.includes("4-6")) + ) { + return env("ANTHROPIC"); + } + if (t.includes("gpt-5.4") || t.includes("gpt-5.5")) return env("OPENAI"); + if (t.includes("glm-5")) return env("ZAI"); + if (t.includes("kimi-k2.5") || t.includes("kimi-k2.6")) return env("MOONSHOT"); + if (t.includes("deepseek-v4")) return env("DEEPSEEK"); + if (t.includes("openai")) return env("OPENAI"); + return undefined; +} + +/** + * Catalog -> preset ModelEntry list (shared by defaultProjectConfig and the server's initial + * config, avoiding duplicate hand-written copies). `provider` and `model_id` are persisted as + * separate fields (`model_id` is the plain upstream id); models whose upstream id can be + * auto-routed by AgentHub leave client_type unset; gateway models (OpenRouter / SiliconFlow) + * explicitly set client_type=openai and inline a preset base_url (no secrets included, so the + * user only needs to supply an API key). + */ +export function presetModelEntries(): ModelEntry[] { + return MODEL_CATALOG.map((m) => ({ + provider: m.provider, + model_id: m.modelId, + ...(m.contextWindow !== undefined ? { context_window: m.contextWindow } : {}), + ...(m.clientType !== undefined ? { client_type: m.clientType } : {}), + ...(m.pricing ? { pricing: { ...m.pricing } } : {}), + // ModelEntry.vision defaults to supported: only models that don't support images + // explicitly persist false (drives the read_image / describe_image choice and input + // image hand-off, see project-config.ts). + ...(m.supportsVision ? {} : { vision: false }), + ...(m.baseUrl !== undefined ? { base_url: m.baseUrl } : {}), + })); +} diff --git a/packages/core/src/state/paths.ts b/packages/core/src/state/paths.ts new file mode 100644 index 0000000..bd81ab2 --- /dev/null +++ b/packages/core/src/state/paths.ts @@ -0,0 +1,114 @@ +/** + * Local directory layout for Agent State and Project config. + * + * Strictly follows the `~/.penguin/data//agents//...` structure. + * This module only provides constants and pure path functions; it never creates directories or reads/writes files. + * Docs: /docs/sessions-and-traces § "Data layout". + */ +import os from "node:os"; +import path from "node:path"; + +/** Default Project id used when none is specified. */ +export const DEFAULT_PROJECT_ID = "default_project"; + +/** Default Agent id used when none is specified. */ +export const DEFAULT_AGENT_ID = "default_agent"; + +/** + * Resolves the local data root directory. + * Prefers the `PENGUIN_HOME` environment variable, otherwise falls back to `~/.penguin/data` + * (under the hidden `~/.penguin` home so it never collides with unrelated folders, and in a + * `data/` subdir kept separate from the installer's binaries under `~/.penguin`). + */ +export function resolveRoot(): string { + return process.env.PENGUIN_HOME ?? path.join(os.homedir(), ".penguin", "data"); +} + +/** `/`. */ +export function projectDir(root: string, projectId: string): string { + return path.join(root, projectId); +} + +/** `/agents`, the container directory holding every Agent in the Project. */ +export function agentsDir(root: string, projectId: string): string { + return path.join(projectDir(root, projectId), "agents"); +} + +/** `/agents/`. */ +export function agentDir(root: string, projectId: string, agentId: string): string { + return path.join(agentsDir(root, projectId), agentId); +} + +/** `/agent_state`. */ +export function agentStateDir(root: string, projectId: string, agentId: string): string { + return path.join(agentDir(root, projectId, agentId), "agent_state"); +} + +/** `/traces`. */ +export function tracesDir(root: string, projectId: string, agentId: string): string { + return path.join(agentDir(root, projectId, agentId), "traces"); +} + +/** `/scratchpad`, the Agent's temporary/draft file directory (the model creates a subdirectory per Session id). */ +export function scratchpadDir(root: string, projectId: string, agentId: string): string { + return path.join(agentDir(root, projectId, agentId), "scratchpad"); +} + +/** `/workspaces`. */ +export function workspacesDir(root: string, projectId: string, agentId: string): string { + return path.join(agentDir(root, projectId, agentId), "workspaces"); +} + +/** + * `/.project_config.toml`, the Project's single config file (a hidden file, not + * shown by default `ls`, written with mode 0600; model entries are inlined with their credential, + * see state/project-config.ts). + */ +export function projectConfigPath(root: string, projectId: string): string { + return path.join(projectDir(root, projectId), ".project_config.toml"); +} + +/** `/system_config.yaml`. */ +export function systemConfigPath(root: string, projectId: string, agentId: string): string { + return path.join(agentStateDir(root, projectId, agentId), "system_config.yaml"); +} + +/** `/AGENTS.md`. */ +export function agentsMdPath(root: string, projectId: string, agentId: string): string { + return path.join(agentStateDir(root, projectId, agentId), "AGENTS.md"); +} + +/** `/.vault.toml`, the Agent-level environment-variable vault (see state/agent-vault.ts). */ +export function agentVaultPath(root: string, projectId: string, agentId: string): string { + return path.join(agentStateDir(root, projectId, agentId), ".vault.toml"); +} + +/** `/tools`, reserved for user-defined Tool config. */ +export function toolsDir(root: string, projectId: string, agentId: string): string { + return path.join(agentStateDir(root, projectId, agentId), "tools"); +} + +/** `/memory`. */ +export function memoryDir(root: string, projectId: string, agentId: string): string { + return path.join(agentStateDir(root, projectId, agentId), "memory"); +} + +/** `/skills`. */ +export function skillsDir(root: string, projectId: string, agentId: string): string { + return path.join(agentStateDir(root, projectId, agentId), "skills"); +} + +/** `/schedule`, the scheduled-task directory (doesn't exist when unconfigured). */ +export function scheduleDir(root: string, projectId: string, agentId: string): string { + return path.join(agentStateDir(root, projectId, agentId), "schedule"); +} + +/** `/benchmarks`, the capability-evaluation question bank and scores (doesn't exist when unconfigured). */ +export function benchmarksDir(root: string, projectId: string, agentId: string): string { + return path.join(agentDir(root, projectId, agentId), "benchmarks"); +} + +/** `/snapshots`, Agent State version snapshots (doesn't exist when unconfigured). */ +export function snapshotsDir(root: string, projectId: string, agentId: string): string { + return path.join(agentDir(root, projectId, agentId), "snapshots"); +} diff --git a/packages/core/src/state/project-config.ts b/packages/core/src/state/project-config.ts new file mode 100644 index 0000000..9dc4b07 --- /dev/null +++ b/packages/core/src/state/project-config.ts @@ -0,0 +1,432 @@ +/** + * Project config storage (`/.project_config.toml`). + * + * Records the available Models, the default Model, and each Model's credential (Model + * is decoupled from Agent — the Model selection isn't stored in Agent State, but maintained by + * the Project). Config is persisted as TOML. + * + * `.project_config.toml` is the Project's **single config file**: a hidden file (not shown by + * default `ls`), written to disk with mode 0600; credentials (api_key / base_url) are **inlined + * on the model entry** rather than split into a supplementary area and a separate secrets file. + * It can only be read/written via the system interfaces (CLI / Web) — never hand-edited by the + * model or the user; the system Prompt is forbidden from reading this file, `loadProjectConfig` + * returns plaintext, and masking is applied at the interface layer (when shown by server / cli). + * + * Model references are **fully split into separate fields**: an entry stores + * `provider` and `model_id` as two independent fields, with the `(provider, model_id)` pair as + * the unique key — string concatenation like `/` is forbidden anywhere in the + * pipeline. `model_id` is the upstream request id, sent to AgentHub unchanged; `default_model` / + * `vision_model` are paired `{ provider, model_id }` references (a TOML inline table). + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { parse as parseToml, stringify as stringifyToml } from "smol-toml"; +import { inferProviderForUpstream, presetModelEntries } from "./model-catalog.js"; +import { projectConfigPath } from "./paths.js"; + +/** Model reference: a `(provider, model_id)` pair (never string-concatenated anywhere). */ +export interface ModelRef { + provider: string; + /** Upstream model id (the request id sent to AgentHub unchanged). */ + model_id: string; +} + +/** + * Display form of a paired reference (shared by error messages and CLI output): + * `(provider=..., model_id=...)`. For display only — it isn't any storage or addressing format. + */ +export function formatModelRef(ref: ModelRef): string { + return `(provider=${ref.provider}, model_id=${ref.model_id})`; +} + +/** + * Pricing for a single Model: three price buckets, in USD per million tokens. + * Docs: /docs/configuration § "Project config". + */ +export interface ModelPricing { + /** Pricing unit tag; currently only `usd_per_mtok` (USD per million tokens). */ + unit: "usd_per_mtok"; + cache_read: number; + cache_write: number; + output: number; +} + +/** + * A single available Model entry (credential inlined, single config file). + * Docs: /docs/models § "The per-Project model table". + */ +export interface ModelEntry { + /** provider group (stored separately from `model_id`; the pair is the entry's unique key). */ + provider: string; + /** Upstream model id: the actual request id sent to AgentHub, used paired with provider for display, pricing, and stats. */ + model_id: string; + context_window?: number; + /** + * AgentHub client protocol (`openai` / `claude-4-8` / `deepseek-v4` / …); defaults to being + * inferred by AgentHub from the request id (`model_id`). A third-party model speaking the + * OpenAI protocol should set this to `openai`. + */ + client_type?: string; + /** + * Display name (the model page card title): only persisted when it differs from the builtin + * catalog (the user renamed it / a custom model); when not persisted, it's inferred from the + * builtin catalog by `(provider, model_id)`, falling back to displaying model_id if it can't be + * inferred. + */ + display_name?: string; + /** + * Whether image input is supported (vision/multimodal); defaults to supported. For a model + * tagged `false` (e.g. DeepSeek): images from conversation input are saved to the session + * scratchpad and handed over as a file path spliced into the text, and the image-reading tool + * switches to describe_image (a vision model reads on its behalf) — the image never directly + * enters that session's history. + */ + vision?: boolean; + /** Pricing info; absent means this Model's cost isn't counted. */ + pricing?: ModelPricing; + /** API key (inlined credential); left empty falls back to the vendor's environment variable. */ + api_key?: string; + /** Custom base URL (inlined credential); preset for gateway models. */ + base_url?: string; + /** api_key's write timestamp (ISO 8601; a display field maintained by the interface layer). */ + created_at?: string; +} + +/** + * Project-level config. + * Docs: /docs/configuration § "Project config". + */ +export interface ProjectConfig { + /** Project display name (the display name is separate from the id, shown as the id when unset). */ + name?: string; + /** Paired reference to the default Model; must point to an entry in `models`. */ + default_model?: ModelRef; + /** + * The vision model used by read_image to read on behalf of a session model (when a session + * model with `vision=false` reads an image, it's handed to this model to describe and the tool + * returns text); must point to an entry in `models` (a paired reference). Unconfigured by + * default — models that don't support images won't be able to read images. + */ + vision_model?: ModelRef; + models: ModelEntry[]; +} + +/** + * Returns the Project's default config: every entry from the preset builtin model catalog + * (including context_window / pricing / vision tags and the preset base_url for gateway models, + * with no keys included) — the user only needs to fill in an API key as needed (left empty falls + * back to the vendor's environment variable). + */ +export function defaultProjectConfig(): ProjectConfig { + return { + default_model: { provider: "deepseek", model_id: "deepseek-v4-pro" }, + models: presetModelEntries(), + }; +} + +/** The old format (concatenated storage id / string reference) is never migrated: reading it reports a clear error immediately (the product hasn't shipped yet). */ +const OLD_FORMAT_HINT = + "产品未发布不做迁移:请删除该配置文件后用 `penguin config model add/default` 重建。"; + +/** Validates the default_model / vision_model fields: must be a { provider, model_id } paired reference. */ +function parseRefField(file: string, name: string, value: unknown): ModelRef | undefined { + if (value === undefined) return undefined; + const ref = value as { provider?: unknown; model_id?: unknown }; + if ( + typeof value !== "object" || + value === null || + typeof ref.provider !== "string" || + typeof ref.model_id !== "string" + ) { + throw new Error( + `.project_config.toml 的 ${name} 是旧版本/非法格式(须为 { provider = "...", model_id = "..." } 成对引用):${file}。${OLD_FORMAT_HINT}`, + ); + } + return { provider: ref.provider, model_id: ref.model_id }; +} + +/** Validates a model entry: both provider and model_id must be strings (an old-format entry is missing provider). */ +function assertModelEntry(file: string, entry: unknown): ModelEntry { + const m = entry as { provider?: unknown; model_id?: unknown }; + if ( + typeof entry !== "object" || + entry === null || + typeof m.provider !== "string" || + typeof m.model_id !== "string" + ) { + throw new Error( + `.project_config.toml 的 models 条目是旧版本/非法格式(provider 与 model_id 须为两个独立字段):${file}。${OLD_FORMAT_HINT}`, + ); + } + return entry as ModelEntry; +} + +/** + * Loads the Project config; returns the default config (without writing to disk) if + * `.project_config.toml` doesn't exist. Returns plaintext (masking is applied at the interface + * layer); reports a clear error when the old format (a string reference / an entry missing + * provider) is read. + */ +export async function loadProjectConfig(root: string, projectId: string): Promise { + const file = projectConfigPath(root, projectId); + let raw: string; + try { + raw = await fs.readFile(file, "utf8"); + } catch (err) { + if ((err as NodeJS.ErrnoException).code === "ENOENT") return defaultProjectConfig(); + throw err; + } + // Defensive: parseToml may return null/undefined for an empty file, and destructuring it would throw a TypeError. + const parsed = (parseToml(raw) ?? {}) as Record; + const defaultModel = parseRefField(file, "default_model", parsed.default_model); + const visionModel = parseRefField(file, "vision_model", parsed.vision_model); + return { + ...(parsed.name !== undefined ? { name: parsed.name as string } : {}), + ...(defaultModel !== undefined ? { default_model: defaultModel } : {}), + ...(visionModel !== undefined ? { vision_model: visionModel } : {}), + models: ((parsed.models as unknown[] | undefined) ?? []).map((m) => assertModelEntry(file, m)), + }; +} + +/** A TOML inline table for a paired reference (reuses smol-toml's string serialization, guaranteeing correct escaping). */ +function tomlInlineRef(ref: ModelRef): string { + const kv = (obj: Record): string => stringifyToml(obj).trim(); + return `{ ${kv({ provider: ref.provider })}, ${kv({ model_id: ref.model_id })} }`; +} + +/** Whether a value has the paired-reference shape ({ provider, model_id }, two string fields). */ +function isModelRefShape(v: unknown): v is ModelRef { + if (v === null || typeof v !== "object" || Array.isArray(v)) return false; + const o = v as Record; + return typeof o.provider === "string" && typeof o.model_id === "string"; +} + +/** + * Renders the full text of `.project_config.toml` — the **single source of the write format + * site-wide** (shared by core's saveProjectConfig and the interface layer's full-table write, to + * avoid the same file ending up in two different formats). + * + * Paired references (default_model / vision_model) are rendered as a TOML inline table + * `{ provider = "...", model_id = "..." }`; `models` is always + * placed last, since any table header after `[[models]]` would be read as its sub-table. Unknown + * extension fields are kept as-is. + */ +export function renderProjectConfigToml(data: Record): string { + const head: string[] = []; + for (const [key, value] of Object.entries(data)) { + if (value === undefined || key === "models") continue; + head.push( + isModelRefShape(value) + ? `${key} = ${tomlInlineRef(value)}` + : stringifyToml({ [key]: value }).trim(), + ); + } + const models = Array.isArray(data.models) ? data.models : []; + return [...head, stringifyToml({ models })].join("\n"); +} + +/** + * Saves the Project config: writes the full table to the single config file + * `.project_config.toml`. The file contains secrets like api_key, so it's written to disk with + * mode 0600 (a hidden file blocks `ls`, not reads; mode only takes effect on creation, so chmod + * converges an existing file too). + */ +export async function saveProjectConfig( + root: string, + projectId: string, + cfg: ProjectConfig, +): Promise { + const file = projectConfigPath(root, projectId); + await fs.mkdir(path.dirname(file), { recursive: true }); + await fs.writeFile(file, renderProjectConfigToml({ ...cfg }), { + encoding: "utf8", + mode: 0o600, + }); + await fs.chmod(file, 0o600); +} + +/** + * Adds or updates a Model: + * - Upserts into `models`, deduplicated by the `(provider, model_id)` pair (provider may be + * omitted — the builtin catalog is used to infer the upstream id's group, falling back to + * custom if it can't be inferred); + * - If `api_key`/`base_url` are provided, they're written inline into the entry; + * - Set as the default Model (a paired reference) when `opts.setDefault` is true. + * Reads the existing config (or the default), saves after the change, and returns the updated + * config. + */ +export async function addModel( + root: string, + projectId: string, + entry: { + /** provider group; inferred from the builtin catalog when omitted (`inferProviderForUpstream`, falling back to custom if it can't be inferred). */ + provider?: string; + /** Upstream model id (sent to AgentHub unchanged). */ + model_id: string; + context_window?: number; + client_type?: string; + /** Whether image input is supported (vision/multimodal); keeps the existing value by default (treated as supported if never set). */ + vision?: boolean; + /** Price input may cover only some buckets; merged and written as a complete `ModelPricing`. */ + pricing?: Partial; + api_key?: string; + base_url?: string; + }, + opts?: { setDefault?: boolean }, +): Promise { + const cfg = await loadProjectConfig(root, projectId); + const provider = entry.provider ?? inferProviderForUpstream(entry.model_id); + + // upsert: layers new fields on top of the existing entry; fields not explicitly provided + // (e.g. context_window) keep their existing value, so a call like "just add an api_key" + // doesn't wipe out the prior config. + const idx = cfg.models.findIndex((m) => m.provider === provider && m.model_id === entry.model_id); + const existing = idx >= 0 ? cfg.models[idx] : undefined; + const modelEntry: ModelEntry = { + provider, + model_id: entry.model_id, + }; + const contextWindow = entry.context_window ?? existing?.context_window; + if (contextWindow !== undefined) { + modelEntry.context_window = contextWindow; + } + const clientType = entry.client_type ?? existing?.client_type; + if (clientType !== undefined) { + modelEntry.client_type = clientType; + } + // The display name and api_key write timestamp are not set by this function; kept as-is on upsert. + if (existing?.display_name !== undefined) { + modelEntry.display_name = existing.display_name; + } + const vision = entry.vision ?? existing?.vision; + if (vision !== undefined) { + modelEntry.vision = vision; + } + // The three price buckets are merged field by field: an unspecified bucket keeps its existing + // value (the same policy as context_window/credential); the unit is fixed to usd_per_mtok, and + // the complete pricing is written as long as any bucket is present. + const mergedPricing: Partial = { + ...existing?.pricing, + ...entry.pricing, + }; + if ( + mergedPricing.cache_read !== undefined || + mergedPricing.cache_write !== undefined || + mergedPricing.output !== undefined + ) { + modelEntry.pricing = { + unit: "usd_per_mtok", + cache_read: mergedPricing.cache_read ?? 0, + cache_write: mergedPricing.cache_write ?? 0, + output: mergedPricing.output ?? 0, + }; + } + // Inline credential entry: fields not provided keep their existing value. + const apiKey = entry.api_key ?? existing?.api_key; + if (apiKey !== undefined) { + modelEntry.api_key = apiKey; + } + const baseUrl = entry.base_url ?? existing?.base_url; + if (baseUrl !== undefined) { + modelEntry.base_url = baseUrl; + } + if (existing?.created_at !== undefined) { + modelEntry.created_at = existing.created_at; + } + if (idx >= 0) { + cfg.models[idx] = modelEntry; + } else { + cfg.models.push(modelEntry); + } + + if (opts?.setDefault) { + cfg.default_model = { provider, model_id: entry.model_id }; + } + + await saveProjectConfig(root, projectId, cfg); + return cfg; +} + +/** + * Sets the default Model and saves. The target reference must exist in `models` (a reference + * pointing outside the config would make createSession error immediately); throws otherwise. + */ +export async function setDefaultModel( + root: string, + projectId: string, + ref: ModelRef, +): Promise { + const cfg = await loadProjectConfig(root, projectId); + if (!getModel(cfg, ref)) { + throw new Error( + `default_model 必须指向已配置的模型:${formatModelRef(ref)} 不在 models 中。用 \`penguin config model list\` 查看已配置的模型。`, + ); + } + cfg.default_model = { provider: ref.provider, model_id: ref.model_id }; + await saveProjectConfig(root, projectId, cfg); + return cfg; +} + +/** + * Sets the vision model used to read images on behalf of read_image, and saves. The target + * reference must exist in `models` and not be tagged `vision=false` (a model that doesn't support + * images can't read on someone's behalf); throws otherwise. + */ +export async function setVisionModel( + root: string, + projectId: string, + ref: ModelRef, +): Promise { + const cfg = await loadProjectConfig(root, projectId); + const entry = getModel(cfg, ref); + if (!entry) { + throw new Error( + `vision_model 必须指向已配置的模型:${formatModelRef(ref)} 不在 models 中。用 \`penguin config model list\` 查看已配置的模型。`, + ); + } + if (entry.vision === false) { + throw new Error(`vision_model 不能指向标注为不支持图片的模型:${formatModelRef(ref)}。`); + } + cfg.vision_model = { provider: ref.provider, model_id: ref.model_id }; + await saveProjectConfig(root, projectId, cfg); + return cfg; +} + +/** Looks up a Model entry exactly by its `(provider, model_id)` paired reference; returns `undefined` if it doesn't exist. */ +export function getModel(cfg: ProjectConfig, ref: ModelRef): ModelEntry | undefined { + return cfg.models.find((m) => m.provider === ref.provider && m.model_id === ref.model_id); +} + +/** + * Resolves a model reference (the **single entry point for "provider omitted"**, shared by core + * and CLI/server — never set up a second one): + * - `provider` given: validated for existence by exact paired reference; + * - `provider` omitted: an exact-match lookup on `model_id` (no fuzzy matching of any kind) — + * resolvable only when **exactly one** entry matches; 0 or multiple matches always report a + * clear error (an ambiguity error lists the candidate paired references). + */ +export function resolveModelRef(cfg: ProjectConfig, modelId: string, provider?: string): ModelRef { + if (provider !== undefined) { + const ref: ModelRef = { provider, model_id: modelId }; + if (!getModel(cfg, ref)) { + throw new Error( + `Model 不在 Project 配置中:${formatModelRef(ref)}。请用 \`penguin config model list\` 查看已配置的模型,或用 \`penguin config model add\` 添加。`, + ); + } + return ref; + } + const candidates = cfg.models.filter((m) => m.model_id === modelId); + if (candidates.length === 1) { + return { provider: candidates[0]!.provider, model_id: modelId }; + } + if (candidates.length === 0) { + throw new Error( + `Model 不在 Project 配置中:没有 model_id 为 ${modelId} 的条目。请用 \`penguin config model list\` 查看已配置的模型,或用 \`penguin config model add\` 添加。`, + ); + } + throw new Error( + `模型引用有歧义:model_id ${modelId} 命中多个条目——${candidates + .map((m) => formatModelRef({ provider: m.provider, model_id: m.model_id })) + .join("、")}。请补充 provider 以给出成对引用。`, + ); +} diff --git a/packages/core/src/trace/index.ts b/packages/core/src/trace/index.ts new file mode 100644 index 0000000..a467cb3 --- /dev/null +++ b/packages/core/src/trace/index.ts @@ -0,0 +1,10 @@ +export { Writer, readTrace } from "./writer.js"; +export type { WriterOptions } from "./writer.js"; +export { + findLatestTraceFile, + latestSessionId, + parseTraceLines, + readTraceTolerant, + resumeTrace, +} from "./resume.js"; +export type { LocatedTraceFile, ResumeResult } from "./resume.js"; diff --git a/packages/core/src/trace/resume.ts b/packages/core/src/trace/resume.ts new file mode 100644 index 0000000..087f3ac --- /dev/null +++ b/packages/core/src/trace/resume.ts @@ -0,0 +1,449 @@ +/** + * Trace replay — the core of Session resume. + * + * Replay produces two results: the **history** injected via setHistory (committed turns only), + * and the **carry-over** input resent with the first `run` after resume. Resume is **best-effort**: + * Trace only records real messages, so synthesized carry-over (`` flattening, pairing + * placeholders) is never written to Trace — replay reconstructs from the original messages + * (unanswered input is resent as-is, pairing placeholders are resynthesized as needed). History is + * guaranteed to be **structurally valid** (turns complete, tool_call pairs matched), not a + * byte-for-byte match of what AgentHub actually received; incomplete model output (thinking/text) + * is allowed to be lost. + * + * Messages are attributed to a Request by **position**, not by content inspection: + * - Input = user-side messages accumulated after the previous `request` `stop` (the first + * Request is `session_meta`) and before this `start` (messages are written to Trace before + * being sent with the request); user-side messages that land between `start` and `stop` + * (output from parallel tools completing during the request) count toward the **next** turn's + * input. + * - Output = assistant messages between this `start` and `stop`. + * + * Determination order: first check file-level compaction closure; then evaluate turn by turn + * (completed turns go to history, others are dropped wholesale while keeping outputs paired with + * already-committed tool_calls); finally, the remaining input is the carry-over, with pairing + * backfill applied. + * Docs: /docs/sessions-and-traces § "Session recovery". + */ +import { readdir, readFile } from "node:fs/promises"; +import { join } from "node:path"; + +import { + emptyTokenCounts, + isCompleteModelMessage, + isEventMessage, + isSessionMeta, + toolCallOutput, + userText, +} from "../omnimessage/index.js"; +import type { + CompactionEndPayload, + CompleteModelMessage, + OmniMessage, + RequestBeginPayload, + RequestEndPayload, + SessionMetaMessage, + TokenCounts, + TokenUsagePayload, + ToolCallPayload, +} from "../omnimessage/index.js"; +import { extractSummary } from "../engine/context-engine.js"; + +/** Replay result: all the state needed to resume a Session. */ +export interface ResumeResult { + /** Committed history (complete model_msg, in order), injected in one shot via setHistory; empty on compaction closure. */ + history: CompleteModelMessage[]; + /** + * Pending input (carry-over): resent alongside new input with the first `run` after resume. + * Already includes pairing-backfill placeholders — placeholders exist only in memory + * (synthesized carry-over is never written to Trace) and are resynthesized on each resume. + */ + carryOver: OmniMessage[]; + /** Compaction closure (file-level): this file's context is fully closed; resume starts a new, empty context. */ + contextClosed: boolean; + /** Compaction closure in summarize mode: the reconstructed `` summary, prepended to the next run's input. */ + pendingSummary?: OmniMessage; + /** Session-level cumulative Token carry-over (the session value from the last token_usage). */ + sessionTokens: TokenCounts; + /** The request.total from the last token_usage (context usage figure). */ + lastRequestTotal: number; + /** Session cumulative turn count carry-over (count of completed requests). */ + sessionTurns: number; + /** Rendering view: this context's complete model_msg plus key event_msg entries (including interrupted turns and their markers); empty on compaction closure. */ + renderMessages: OmniMessage[]; + /** The file's first session_meta; null if missing (unresumable — the caller reports the error). */ + meta: SessionMetaMessage | null; +} + +/** Content of the pairing-backfill placeholder output (the tool hadn't finished and no output was persisted before the process exited). */ +const PROCESS_EXIT_PLACEHOLDER = "[interrupted: process exited before the tool finished]"; + +/** + * Parse Trace JSONL content. Tolerates a **truncated last line** left behind by an abnormal + * process exit (that line is ignored); corruption in the middle is outside the crash window + * (append-only, single writer), so it throws loudly. + */ +export function parseTraceLines(content: string): OmniMessage[] { + const lines = content.split("\n"); + const out: OmniMessage[] = []; + for (let i = 0; i < lines.length; i++) { + const line = lines[i]!.trim(); + if (!line) continue; + try { + out.push(JSON.parse(line) as OmniMessage); + } catch (err) { + const isLastNonEmpty = lines.slice(i + 1).every((l) => l.trim().length === 0); + if (isLastNonEmpty) break; + throw err; + } + } + return out; +} + +/** Read and parse a Trace file (tolerates a truncated last line). */ +export async function readTraceTolerant(path: string): Promise { + return parseTraceLines(await readFile(path, "utf8")); +} + +/** A located Trace file: its path, containing date-directory name, and index. */ +export interface LocatedTraceFile { + path: string; + dateDir: string; + index: number; +} + +const TRACE_FILE_RE = /^(.+)_(\d{3})\.jsonl$/; + +/** + * Locate the **highest-index** Trace file for a Session (one Trace file corresponds to one + * complete model context). Scans `//_.jsonl`; returns + * null if no match is found. + */ +export async function findLatestTraceFile( + tracesDir: string, + sessionId: string, +): Promise { + let best: LocatedTraceFile | null = null; + for (const dateDir of await listDirs(tracesDir)) { + for (const file of await listFiles(join(tracesDir, dateDir))) { + const match = TRACE_FILE_RE.exec(file); + if (!match || match[1] !== sessionId) continue; + const index = Number(match[2]); + if (!best || index > best.index) { + best = { path: join(tracesDir, dateDir, file), dateDir, index }; + } + } + } + return best; +} + +/** + * The id of the most recent Session under this Agent, determined by the timestamp embedded in + * session_id (ids are zero-padded, so lexical order equals chronological order). Returns null if + * there are no Sessions. + */ +export async function latestSessionId(tracesDir: string): Promise { + const dateDirs = (await listDirs(tracesDir)).sort((a, b) => b.localeCompare(a)); + for (const dateDir of dateDirs) { + const files = (await listFiles(join(tracesDir, dateDir))).sort((a, b) => b.localeCompare(a)); + for (const file of files) { + const match = TRACE_FILE_RE.exec(file); + if (!match) continue; + if (await hasResumableTraceContent(join(tracesDir, dateDir, file))) { + return match[1]!; + } + } + } + return null; +} + +async function hasResumableTraceContent(file: string): Promise { + try { + const messages = await readTraceTolerant(file); + return messages.some((msg) => !isSessionMeta(msg)); + } catch { + return false; + } +} + +async function listDirs(dir: string): Promise { + try { + const entries = await readdir(dir, { withFileTypes: true }); + return entries.filter((e) => e.isDirectory()).map((e) => e.name); + } catch { + return []; // traces directory doesn't exist yet: no Sessions + } +} + +async function listFiles(dir: string): Promise { + try { + const entries = await readdir(dir, { withFileTypes: true }); + return entries.filter((e) => e.isFile()).map((e) => e.name); + } catch { + return []; + } +} + +function isRequestBegin(msg: OmniMessage): msg is OmniMessage { + return isEventMessage(msg) && (msg.payload as { type?: string }).type === "request_begin"; +} + +function isRequestEnd(msg: OmniMessage): msg is OmniMessage { + return isEventMessage(msg) && (msg.payload as { type?: string }).type === "request_end"; +} + +function isCompactionEnd(msg: OmniMessage): msg is OmniMessage { + return isEventMessage(msg) && (msg.payload as { type?: string }).type === "compaction_end"; +} + +function toolCallOutputId(msg: OmniMessage): string | null { + const p = msg.payload as { type?: string; tool_call_id?: string }; + return p.type === "tool_call_output" ? (p.tool_call_id ?? null) : null; +} + +/** + * Replay a Trace file (the current context), reconstructing history and carry-over input. + * Input is the message sequence parsed by `readTraceTolerant`. + */ +export function resumeTrace(messages: OmniMessage[]): ResumeResult { + const meta = (messages.find(isSessionMeta) as SessionMetaMessage | undefined) ?? null; + const sessionTokens = lastSessionTokens(messages); + const lastRequestTotal = lastRequestTotalOf(messages); + + // —— First check the file-level case: compaction closure (the file ends with a completed + // compaction stop and no new file was opened, i.e. it's still the latest index at resume time) + // — this file's context is fully closed, so the whole file is not replayed. + const last = messages[messages.length - 1]; + if (last && isCompactionEnd(last)) { + const p = last.payload; + if (p.status === "completed") { + const result: ResumeResult = { + history: [], + carryOver: [], + contextClosed: true, + sessionTokens, + lastRequestTotal: 0, // new context has no usage yet + sessionTurns: 0, // turn count resets after compaction completes + renderMessages: [], + meta, + }; + if (p.mode === "summarize") { + // Reconstruct the summary from the compaction request's output (the assistant text of + // the last completed Request). + const summaryText = lastCompletedRequestText(messages); + result.pendingSummary = userText( + `\n${extractSummary(summaryText)}\n`, + ); + } + return result; + } + } + + // —— Turn-by-turn determination + pending-input convergence. + const history: CompleteModelMessage[] = []; + /** Pending-input buffer: user-side messages not yet sent with any committed Request. */ + let pending: OmniMessage[] = []; + /** The current Request's input snapshot (frozen at begin) and its outputs. */ + let snapshot: OmniMessage[] = []; + let outputs: CompleteModelMessage[] = []; + let inRequest = false; + /** Whether we're between a matched pair of compaction events: the compaction prompt in this + * span is not conversational input and must not be resent as-is if uncommitted. */ + let inCompaction = false; + /** Ids of tool_calls that are committed (in history) and ids of outputs that are paired (in + * history input). */ + const committedCallIds = new Set(); + const answeredIds = new Set(); + let sessionTurns = 0; + const renderMessages: OmniMessage[] = []; + + const placeholderFor = (id: string): CompleteModelMessage => + toolCallOutput({ + output: PROCESS_EXIT_PLACEHOLDER, + toolCallId: id, + stopReason: "aborted", + }) as CompleteModelMessage; + + const dropUncommittedRound = (): void => { + // Uncommitted turn: the whole turn is excluded from history. Its **original input** (user + // text/images and structured tool output) goes back into the pending buffer as-is — best + // effort to resend "the last input that got no response"; incomplete model output + // (thinking/text) is allowed to be lost. + // Exception: a failed compaction turn's compaction prompt is not conversational input and is + // not reclaimed (structured output is still reclaimed, subject to eligibility filtering). + const keep = inCompaction ? snapshot.filter((m) => toolCallOutputId(m) !== null) : snapshot; + pending = [...keep, ...pending]; + snapshot = []; + outputs = []; + inRequest = false; + }; + + for (const msg of messages) { + if (isSessionMeta(msg)) continue; + + if (isRequestBegin(msg)) { + // Defensive: the previous turn had begin but no end (shouldn't happen mid-file — a process + // exit only affects the tail) — treat it as uncommitted. + if (inRequest) dropUncommittedRound(); + inRequest = true; + snapshot = pending; + pending = []; + continue; + } + if (isRequestEnd(msg)) { + if (msg.payload.status === "completed") { + // Structural eligibility filter (same rule as the final carry-over): a tool_call_output + // in the snapshot is kept only if it pairs with a tool_call that is **committed and not + // yet answered** — tool output dispatched by a dropped turn gets persisted in the next + // turn's input span, but its tool_call isn't in history, so keeping it as-is would create + // an orphan tool_result with no preceding tool_use, which every request would be rejected + // for by the provider after resume ("Replay Rules"' strict pairing guarantee). + const eligible = snapshot.filter((m) => { + const id = toolCallOutputId(m); + return id === null || (committedCallIds.has(id) && !answeredIds.has(id)); + }); + // Structural repair (best-effort): a tool_call committed earlier but still unpaired — its + // matching output was once sent as synthesized carry-over but **never written to Trace** + // — resynthesize a placeholder before this turn's input, to keep the injected history + // structurally valid (every assistant tool_use is followed by a user tool_result). + const snapshotOutputIds = new Set( + eligible.map(toolCallOutputId).filter((id): id is string => id !== null), + ); + for (const id of committedCallIds) { + if (answeredIds.has(id) || snapshotOutputIds.has(id)) continue; + history.push(placeholderFor(id)); + answeredIds.add(id); + } + history.push(...(eligible as CompleteModelMessage[]), ...outputs); + for (const id of snapshotOutputIds) answeredIds.add(id); + for (const m of outputs) { + const p = m.payload as Partial; + if (p.type === "tool_call" && p.stop_reason === "completed" && p.tool_call_id) { + committedCallIds.add(p.tool_call_id); + } + } + sessionTurns += 1; + snapshot = []; + outputs = []; + inRequest = false; + } else { + dropUncommittedRound(); + } + continue; + } + + if (isCompleteModelMessage(msg)) { + renderMessages.push(msg); + const role = (msg.payload as { role?: string }).role; + if (role === "user") { + // All user-side messages go into the pending buffer: ones between begin/end (output from + // parallel tools completing during the request) count toward the next turn's input. + pending.push(msg); + } else if (inRequest) { + outputs.push(msg); + } + // Defensive: assistant messages outside a span (shouldn't happen) are excluded from + // history, kept only for rendering. + continue; + } + if (isEventMessage(msg)) { + const t = (msg.payload as { type?: string }).type; + if (t === "compaction_begin") inCompaction = true; + else if (t === "compaction_end") inCompaction = false; + else if (t === "abort" && (msg.origin?.length ?? 0) === 0) renderMessages.push(msg); + // Other events (token_usage / approval_decision, etc.) don't participate in turn + // determination. + } + } + + // File ends mid-request (begin but no end — the process exited during a request): treat as an + // uncommitted turn. + if (inRequest) dropUncommittedRound(); + + // —— Structured resend eligibility filter: a tool_call_output in the pending input is kept + // only if it pairs with a tool_call that is **committed and not yet answered**. Tool output + // dispatched by an uncommitted turn itself (which may be persisted before or after that turn's + // end — the tool and LLM streams run concurrently, so finishing order is unpredictable) is + // dropped entirely: its tool_call isn't in history, so a structured resend would form an orphan + // tool_result with no preceding tool_use, which the provider would reject. + pending = pending.filter((m) => { + const id = toolCallOutputId(m); + return id === null || (committedCallIds.has(id) && !answeredIds.has(id)); + }); + + // —— Pairing backfill: for any tool_call committed in history with no matching output in + // either history or pending input, add an interrupted-state placeholder to pending input. + // Placeholders exist only in memory (synthesized carry-over is never written to Trace) and are + // resynthesized on each resume as needed. + const pairedIds = new Set(answeredIds); + for (const m of pending) { + const id = toolCallOutputId(m); + if (id !== null) pairedIds.add(id); + } + const pairingBackfill: OmniMessage[] = []; + for (const id of committedCallIds) { + if (pairedIds.has(id)) continue; + pairingBackfill.push(placeholderFor(id)); + } + + return { + history, + carryOver: [...pending, ...pairingBackfill], + contextClosed: false, + sessionTokens, + lastRequestTotal, + sessionTurns, + renderMessages, + meta, + }; +} + +/** The session cumulative total from the last token_usage (zero if none). */ +function lastSessionTokens(messages: OmniMessage[]): TokenCounts { + for (let i = messages.length - 1; i >= 0; i--) { + const msg = messages[i]!; + if (!isEventMessage(msg)) continue; + const p = msg.payload as Partial; + if (p.type === "token_usage" && p.session) return p.session; + } + return emptyTokenCounts(); +} + +/** The request.total from the last token_usage (context usage figure; 0 if none). */ +function lastRequestTotalOf(messages: OmniMessage[]): number { + for (let i = messages.length - 1; i >= 0; i--) { + const msg = messages[i]!; + if (!isEventMessage(msg)) continue; + const p = msg.payload as Partial; + if (p.type === "token_usage" && p.request) return p.request.total; + } + return 0; +} + +/** Concatenated assistant text of the last completed Request (on compaction closure, this is the compaction request's output). */ +function lastCompletedRequestText(messages: OmniMessage[]): string { + let text = ""; + let current = ""; + let inRequest = false; + for (const msg of messages) { + if (isRequestBegin(msg)) { + inRequest = true; + current = ""; + continue; + } + if (isRequestEnd(msg)) { + { + // A completed request with empty text still overwrites (we take the text of “the last + // completed request”, even if empty) — otherwise a textless compaction output would fall + // back to an earlier turn's normal reply and get mistakenly injected as the summary; the + // in-process path yields an empty summary here (extractSummary(“”)). + if (msg.payload.status === "completed") text = current; + inRequest = false; + } + continue; + } + if (!inRequest || !isCompleteModelMessage(msg)) continue; + const p = msg.payload as { type?: string; role?: string; text?: string }; + if (p.type === "text" && p.role === "assistant" && p.text) current += p.text; + } + return text; +} diff --git a/packages/core/src/trace/writer.ts b/packages/core/src/trace/writer.ts new file mode 100644 index 0000000..6df78ba --- /dev/null +++ b/packages/core/src/trace/writer.ts @@ -0,0 +1,149 @@ +/** + * Trace writer — append-only JSON Lines. + * + * Docs: packages/docs/content/sessions-and-traces.{zh,en}.md (site path + * /docs/sessions-and-traces) documents the file layout and recording rules. + * + * Design points: + * - Every observable action is appended to Trace; historical events are never modified in place + * (append-only). + * - One Trace file corresponds to one complete model context; when the context is compacted + * and a new segment is produced, `rotate()` starts a new, separately numbered file. + * - Only "recordable" messages are written: `session_meta`, complete `model_msg`, and all + * `event_msg`; streaming `partial_*` messages are skipped (the producer appends the + * corresponding complete message once the segment ends); nested child-session messages are + * never written (their spawn location is recorded via the `subagent` pointer event written by + * context_engine). + * - Path convention: `//_.jsonl`. + */ +import { appendFile, mkdir, readFile } from "node:fs/promises"; +import { dirname, join } from "node:path"; + +import { + PartialAggregator, + isCompleteModelMessage, + isEventMessage, + isSessionMeta, +} from "../omnimessage/index.js"; +import type { OmniMessage } from "../omnimessage/index.js"; +import { formatLocalDate } from "../internal/dates.js"; + +export interface WriterOptions { + /** Trace root directory, typically `/traces`. */ + tracesDir: string; + /** Current Session id, written into the file name. */ + sessionId: string; + /** The time used to derive the date subdirectory; defaults to `new Date()`. */ + date?: Date; + /** + * Directly specifies the date subdirectory name (used when Session resumption continues + * writing to the original file: the Trace file follows the context, not the date); takes + * priority over `date`. + */ + dateDir?: string; + /** Starting Trace index (used when Session resumption continues the original index); defaults to 1. */ + startIndex?: number; +} + +/** Zero-pads a Trace index to 3 digits, e.g. 1 -> "001". */ +function formatIndex(index: number): string { + return index.toString().padStart(3, "0"); +} + +/** + * Determines whether an OmniMessage should be written to Trace (skips streaming partial_* and nested child-session messages). + * + * Child-session messages are never written to this Trace: the child Session has its own complete + * Trace, and recording it again would distort this Trace's statistics. The spawn location is + * recorded via the `subagent` pointer event (recording only the child Session id) that + * context_engine writes at the spawn site; when the session is reopened, the server uses this to + * re-attach the child session to its corresponding run_subagent tool card. + * Docs: /docs/sessions-and-traces § "Trace design". + */ +function isRecordable(msg: OmniMessage): boolean { + if (msg.origin && msg.origin.length > 0) return false; + return isCompleteModelMessage(msg) || isEventMessage(msg) || isSessionMeta(msg); +} + +/** + * append-only JSONL Trace writer. + * + * Single-writer scenario (MVP): concurrency safety isn't required, but every write uses + * `appendFile` (O_APPEND) rather than caching a file handle and seeking to write, avoiding + * overwriting existing content; this also removes the need for an explicit close. + */ +export class Writer { + private readonly tracesDir: string; + private readonly sessionId: string; + private readonly dateDir: string; + /** Current Trace index, starting at 1; incremented by `rotate()`. */ + private index = 1; + /** Set true once the date directory has been created for the current file, to avoid a redundant mkdir. */ + private ensuredDirForIndex = -1; + + constructor(opts: WriterOptions) { + this.tracesDir = opts.tracesDir; + this.sessionId = opts.sessionId; + this.dateDir = opts.dateDir ?? formatLocalDate(opts.date ?? new Date()); + this.index = opts.startIndex ?? 1; + } + + /** Absolute path of the current Trace file. */ + currentPath(): string { + const fileName = `${this.sessionId}_${formatIndex(this.index)}.jsonl`; + return join(this.tracesDir, this.dateDir, fileName); + } + + /** + * Appends one message. Only written if it's a recordable message; streaming `partial_*` is + * skipped. `mkdir -p`s the date directory on the first write to the current file. + */ + async write(msg: OmniMessage): Promise { + if (!isRecordable(msg)) return; + const path = this.currentPath(); + if (this.ensuredDirForIndex !== this.index) { + await mkdir(dirname(path), { recursive: true }); + this.ensuredDirForIndex = this.index; + } + await appendFile(path, `${JSON.stringify(msg)}\n`, "utf8"); + } + + /** Writes multiple messages in sequence. */ + async writeAll(msgs: OmniMessage[]): Promise { + for (const msg of msgs) { + await this.write(msg); + } + } + + /** + * Aggregates a message stream mixed with streaming `partial_*` into complete messages first, + * then writes them per the `write` convention. A convenience helper: `write` skips partial_* + * by default (the producer will already append the complete message), so this method is only + * needed when reconstructing a complete context from raw streaming fragments. + */ + async aggregateAndWrite(msgs: OmniMessage[]): Promise { + const agg = new PartialAggregator(); + for (const msg of msgs) { + await this.writeAll(agg.push(msg)); + } + await this.writeAll(agg.flush()); + } + + /** + * Starts a new Trace file: increments the index, so the next `write` goes to the new file. + * Used to split into a separate file when the context is compacted and a new context segment is produced. + * Docs: /docs/sessions-and-traces § "Trace design". + */ + async rotate(): Promise { + this.index += 1; + } +} + +/** Parses a Trace file line by line (ignoring blank lines), for testing and later reads. */ +export async function readTrace(path: string): Promise { + const content = await readFile(path, "utf8"); + return content + .split("\n") + .filter((line) => line.trim().length > 0) + .map((line) => JSON.parse(line) as OmniMessage); +} diff --git a/packages/core/test/agent-lifecycle.test.ts b/packages/core/test/agent-lifecycle.test.ts new file mode 100644 index 0000000..a8f4a76 --- /dev/null +++ b/packages/core/test/agent-lifecycle.test.ts @@ -0,0 +1,105 @@ +/** + * Agent lifecycle: new session id format + no more `.penguin` symlink in the Workspace. + * + * - sessionId looks like `session-YYYY-MM-DD-HH-mm-ss-<8-digit hex>` (local time, zero-padded fields). + * - createSession no longer creates any `.penguin` symlink inside the Workspace, nor touches existing + * Workspace files; the model reaches Agent State and other absolute paths directly by combining the + * Project Dir / Agent ID placeholders from the system prompt. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { createAgent } from "../src/index.js"; +import { formatSessionId } from "../src/internal/session-support.js"; +import { projectDir } from "../src/state/paths.js"; +import { stubProviderKeys } from "./provider-keys.js"; + +let tmpRoot: string; +let prevHome: string | undefined; +let restoreKeys: () => void; + +beforeEach(async () => { + prevHome = process.env.PENGUIN_HOME; + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-harness-lifecycle-")); + process.env.PENGUIN_HOME = tmpRoot; + restoreKeys = stubProviderKeys(); +}); + +afterEach(async () => { + if (prevHome === undefined) delete process.env.PENGUIN_HOME; + else process.env.PENGUIN_HOME = prevHome; + restoreKeys(); + await fs.rm(tmpRoot, { recursive: true, force: true }); +}); + +const SESSION_ID_RE = /^session-\d{4}-\d{2}-\d{2}-\d{2}-\d{2}-\d{2}-[0-9a-f]{8}$/; + +describe("formatSessionId", () => { + it("matches the session-YYYY-MM-DD-HH-mm-ss-<8hex> format", () => { + expect(formatSessionId()).toMatch(SESSION_ID_RE); + }); + + it("uses local time fields with zero padding", () => { + // 2026-06-19 15:28:08 local time -> session-2026-06-19-15-28-08-. + const d = new Date(2026, 5, 19, 15, 28, 8); + expect(formatSessionId(d)).toMatch(/^session-2026-06-19-15-28-08-[0-9a-f]{8}$/); + }); + + it("generates distinct ids on repeated calls", () => { + expect(formatSessionId()).not.toBe(formatSessionId()); + }); +}); + +describe("Agent.createSession session id + no .penguin symlink", () => { + it("assigns a sessionId in the new format", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws"); + await fs.mkdir(ws, { recursive: true }); + const session = await agent.createSession({ workspaceDir: ws }); + expect(session.sessionId).toMatch(SESSION_ID_RE); + }); + + it("does not create a .penguin entry in the workspace", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws"); + await fs.mkdir(ws, { recursive: true }); + await agent.createSession({ workspaceDir: ws }); + await expect(fs.lstat(path.join(ws, ".penguin"))).rejects.toThrow(); + // agent_state also no longer creates traces/notes symlinks. + await expect(fs.lstat(path.join(agent.state.stateDir, "traces"))).rejects.toThrow(); + await expect(fs.lstat(path.join(agent.state.stateDir, "notes"))).rejects.toThrow(); + }); + + it("leaves a user's pre-existing .penguin file untouched", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws"); + await fs.mkdir(ws, { recursive: true }); + const linkPath = path.join(ws, ".penguin"); + await fs.writeFile(linkPath, "user-data", "utf8"); + await agent.createSession({ workspaceDir: ws }); + expect(await fs.readFile(linkPath, "utf8")).toBe("user-data"); + }); + + it("is idempotent: repeated createSession registers no exit listeners", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws"); + await fs.mkdir(ws, { recursive: true }); + const before = process.listenerCount("exit"); + for (let i = 0; i < 12; i++) { + await agent.createSession({ workspaceDir: ws }); + } + expect(process.listenerCount("exit") - before).toBe(0); + }); + + it("injects Project Dir and Agent ID into the assembled system prompt", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws"); + await fs.mkdir(ws, { recursive: true }); + const session = await agent.createSession({ workspaceDir: ws }); + const prompt = (session.metaMessage.payload as { system_prompt: string }).system_prompt; + expect(prompt).toContain(`Agent ID: ${agent.state.agentId}`); + expect(prompt).toContain(`Project Dir: ${projectDir(tmpRoot, agent.state.projectId)}`); + expect(prompt).not.toContain(".penguin"); + }); +}); diff --git a/packages/core/test/agent-skills.test.ts b/packages/core/test/agent-skills.test.ts new file mode 100644 index 0000000..b88ec92 --- /dev/null +++ b/packages/core/test/agent-skills.test.ts @@ -0,0 +1,201 @@ +/** + * On-disk behavior of an Agent's installed Skills: installSkill / + * removeSkill / listInstalledSkills, and metadata injection via skillMetadataSection / + * assembleSystemPrompt. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { librarySkill } from "@prismshadow/penguin-skills"; +import { + AGENTS_MD_PLACEHOLDER, + DEFAULT_AGENT_ID, + DEFAULT_PROJECT_ID, + SKILL_METADATA_PLACEHOLDER, + agentStateDir, + assembleSystemPrompt, + installSkill, + listInstalledSkills, + removeSkill, + skillMetadataSection, + skillsDir, +} from "../src/state/index.js"; + +let tmpRoot: string; + +beforeEach(async () => { + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-skills-")); +}); + +afterEach(async () => { + await fs.rm(tmpRoot, { recursive: true, force: true }); +}); + +const install = (name: string, content: string, icon?: string) => + installSkill(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, { + name, + content, + ...(icon !== undefined ? { icon } : {}), + }); +const list = () => listInstalledSkills(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID); +const skillFile = (name: string, file: string) => + path.join(skillsDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), name, file); +const skillMd = (name: string) => skillFile(name, "SKILL.md"); +const skillIcon = (name: string) => skillFile(name, "icon.svg"); + +describe("installSkill / removeSkill", () => { + it("writes skills//SKILL.md verbatim with a trailing newline", async () => { + const skill = librarySkill("penguin-cli")!; + await install(skill.name, skill.content); + expect(await fs.readFile(skillMd("penguin-cli"), "utf8")).toBe(skill.content); + + // Content without a trailing newline gets one appended; reinstalling overwrites. + await install("penguin-cli", "---\nname: penguin-cli\nversion: 2\n---\n\nNew body"); + expect(await fs.readFile(skillMd("penguin-cli"), "utf8")).toBe( + "---\nname: penguin-cli\nversion: 2\n---\n\nNew body\n", + ); + expect((await list()).map((s) => s.version)).toEqual([2]); + }); + + it("writes icon.svg alongside SKILL.md, and reinstalling without icon removes it", async () => { + // A library skill with an icon: installing writes it to disk alongside SKILL.md. + const skill = librarySkill("penguin-sdk")!; + expect(skill.icon).toBeTruthy(); + await install(skill.name, skill.content, skill.icon); + expect(await fs.readFile(skillIcon("penguin-sdk"), "utf8")).toBe(skill.icon); + + // Overwrite semantics: this install has no icon -> the old icon.svg is removed, and the + // directory matches this install's content exactly. + await install("penguin-sdk", "---\nname: penguin-sdk\nversion: 2\n---\n\nNew body\n"); + await expect(fs.access(skillIcon("penguin-sdk"))).rejects.toThrow(); + expect(await fs.readFile(skillMd("penguin-sdk"), "utf8")).toContain("New body"); + }); + + it("rejects invalid skill names (path traversal safety)", async () => { + await expect(install("../evil", "x")).rejects.toThrow(/skill_name/); + await expect(install("a/b", "x")).rejects.toThrow(/skill_name/); + await expect(removeSkill(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "..")).rejects.toThrow( + /skill_name/, + ); + }); + + it("removeSkill deletes the whole skill directory and is idempotent", async () => { + const skill = librarySkill("penguin-sdk")!; + await install(skill.name, skill.content); + await removeSkill(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "penguin-sdk"); + await expect(fs.access(skillMd("penguin-sdk"))).rejects.toThrow(); + // Idempotent when it no longer exists: does not throw. + await removeSkill(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "penguin-sdk"); + expect(await list()).toEqual([]); + }); +}); + +describe("listInstalledSkills", () => { + it("returns [] when the skills directory does not exist", async () => { + expect(await list()).toEqual([]); + }); + + it("parses frontmatter and sorts by name", async () => { + await install( + "zeta", + "---\nname: zeta\ndescription: Z skill.\nversion: 3\nupdated: 2026-07-16\n---\n\nBody\n", + ); + await install( + "alpha", + "---\nname: alpha\ndescription: A skill.\nversion: 1\nupdated: 2026-07-16\n---\n\nBody\n", + ); + expect(await list()).toEqual([ + { name: "alpha", description: "A skill.", version: 1, updated: "2026-07-16" }, + { name: "zeta", description: "Z skill.", version: 3, updated: "2026-07-16" }, + ]); + }); + + it("returns icon.svg content and passes short description fields through", async () => { + const icon = '\n'; + await install( + "with-extras", + "---\nname: with-extras\ndescription: Long description here.\nshort_description: Short one.\nshort_description_zh: 短描述。\nversion: 1\nupdated: 2026-07-17\n---\n\nBody\n", + icon, + ); + await install( + "plain", + "---\nname: plain\ndescription: Plain skill.\nversion: 1\nupdated: 2026-07-17\n---\n\nBody\n", + ); + const skills = await list(); + expect(skills).toEqual([ + { name: "plain", description: "Plain skill.", version: 1, updated: "2026-07-17" }, + { + name: "with-extras", + description: "Long description here.", + shortDescription: "Short one.", + shortDescriptionZh: "短描述。", + version: 1, + updated: "2026-07-17", + icon, + }, + ]); + // Entries missing icon / short description omit the corresponding field (undefined does + // not produce a key; the interface layer's conditional spread relies on this convention). + expect("icon" in skills[0]!).toBe(false); + expect("shortDescription" in skills[0]!).toBe(false); + }); + + it("uses the directory name as the skill identity even when frontmatter name disagrees", async () => { + // A hand-written or network-sourced skill may have a frontmatter name that differs from + // its directory name: the directory name is the addressing key used for install, uninstall, + // and Prompt lookup, so the listing must follow it (frontmatter fields are display-only). + await install( + "local-name", + "---\nname: upstream-name\ndescription: Fetched skill.\nversion: 2\nupdated: 2026-07-01\n---\n\nBody\n", + ); + expect(await list()).toEqual([ + { name: "local-name", description: "Fetched skill.", version: 2, updated: "2026-07-01" }, + ]); + }); + + it("falls back to directory-name metadata for broken frontmatter and skips non-skill entries", async () => { + // A SKILL.md without frontmatter: falls back to directory name + empty description + version 1. + await install("broken", "# No frontmatter here\n"); + // Directories without a SKILL.md, and stray files, do not count as Skills. + const dir = skillsDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID); + await fs.mkdir(path.join(dir, "empty-dir"), { recursive: true }); + await fs.writeFile(path.join(dir, "stray.md"), "stray", "utf8"); + expect(await list()).toEqual([{ name: "broken", description: "", version: 1, updated: "" }]); + }); +}); + +describe("skillMetadataSection / assembleSystemPrompt 注入", () => { + it("renders one `- \\`name\\` — description` line per skill; empty input renders empty", () => { + expect(skillMetadataSection([])).toBe(""); + expect( + skillMetadataSection([ + { name: "a", description: "Does A.", version: 1, updated: "2026-07-16" }, + { name: "b", description: "", version: 1, updated: "" }, + ]), + ).toBe("- `a` — Does A.\n- `b`"); + }); + + it("replaces {{SKILL_METADATA}} with metadata lines, or an empty string when absent", () => { + const state = { + root: tmpRoot, + projectId: DEFAULT_PROJECT_ID, + agentId: DEFAULT_AGENT_ID, + stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + systemConfig: { + system_prompt: ["before", AGENTS_MD_PLACEHOLDER, SKILL_METADATA_PLACEHOLDER, "after"].join( + "\n", + ), + }, + agentsMd: "# Agent Rules", + }; + const prompt = assembleSystemPrompt(state, undefined, undefined, [ + { name: "demo", description: "Demo skill.", version: 1, updated: "2026-07-16" }, + ]); + expect(prompt).toBe(["before", "# Agent Rules", "- `demo` — Demo skill.", "after"].join("\n")); + // Not provided / empty list: the placeholder is replaced with an empty string, no residue left. + const empty = assembleSystemPrompt(state); + expect(empty).toBe(["before", "# Agent Rules", "", "after"].join("\n")); + expect(empty).not.toContain(SKILL_METADATA_PLACEHOLDER); + }); +}); diff --git a/packages/core/test/agent.test.ts b/packages/core/test/agent.test.ts new file mode 100644 index 0000000..f4f9203 --- /dev/null +++ b/packages/core/test/agent.test.ts @@ -0,0 +1,256 @@ +/** + * Agent.createSession's Workspace handling and vault injection (no network needed; only + * constructs the Session, never sends a request). + * + * Regression: an explicitly given Workspace must be an existing directory. When it + * does not exist, a clear error must be thrown rather than auto-creating it, and bash must not + * be started with an invalid cwd after Session creation, which would throw a misleading + * `spawn bash ENOENT`. A temp directory is only created when no Workspace is specified. + * + * vault: the Agent vault's (agent_state/.vault.toml) **key names** are + * injected into the assembled system prompt; values are never injected. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { + addModel, + createAgent, + DEFAULT_AGENT_ID, + DEFAULT_PROJECT_ID, + installSkill, + setVaultEntry, +} from "../src/index.js"; +import { effectiveMaxContextLength } from "../src/agent.js"; +import { stubProviderKeys } from "./provider-keys.js"; + +let tmpRoot: string; +let prevHome: string | undefined; +let restoreKeys: () => void; + +beforeEach(async () => { + prevHome = process.env.PENGUIN_HOME; + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-harness-")); + process.env.PENGUIN_HOME = tmpRoot; + restoreKeys = stubProviderKeys(); +}); + +afterEach(async () => { + if (prevHome === undefined) delete process.env.PENGUIN_HOME; + else process.env.PENGUIN_HOME = prevHome; + restoreKeys(); + await fs.rm(tmpRoot, { recursive: true, force: true }); +}); + +describe("effectiveMaxContextLength (压缩阈值按模型窗口钳制)", () => { + it("clamps to 75% of a small model window; leaves big/unknown windows and off untouched", () => { + expect(effectiveMaxContextLength(128000, 32768)).toBe(24576); // small window: clamp to 75% + expect(effectiveMaxContextLength(128000, 200000)).toBe(128000); // ample window: unchanged + expect(effectiveMaxContextLength(-1, 32768)).toBe(-1); // off: no clamping + expect(effectiveMaxContextLength(0, 32768)).toBe(0); // off: no clamping + expect(effectiveMaxContextLength(128000, "unknown")).toBe(128000); // unknown window: no clamping + }); +}); + +describe("Agent.createSession workspace handling", () => { + it("throws a clear error when the given workspace does not exist (no auto-create)", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "nested", "does-not-exist"); + await expect(agent.createSession({ workspaceDir: ws })).rejects.toThrow(/不存在/); + // Must not be auto-created. + await expect(fs.stat(ws)).rejects.toThrow(); + }); + + it("throws when the given workspace path is not a directory", async () => { + const agent = await createAgent(); + const filePath = path.join(tmpRoot, "a-file"); + await fs.writeFile(filePath, "x", "utf8"); + await expect(agent.createSession({ workspaceDir: filePath })).rejects.toThrow(/不是目录/); + }); + + it("accepts an existing directory and resolves it to an absolute path", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws"); + await fs.mkdir(ws, { recursive: true }); + const session = await agent.createSession({ workspaceDir: ws }); + expect(session.workspaceDir).toBe(ws); + expect(path.isAbsolute(session.workspaceDir)).toBe(true); + }); + + it("rejects a modelId that is not in the Project config with a clear error", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws-bad-model"); + await fs.mkdir(ws, { recursive: true }); + // A reference outside the config is not silently allowed (the unique key is provider + + // model_id); the error is thrown before creating the temp Workspace. + await expect( + agent.createSession({ workspaceDir: ws, modelId: "not-configured-model" }), + ).rejects.toThrow(/不在 Project 配置中/); + await expect( + agent.createSession({ workspaceDir: ws, modelId: "deepseek-v4-pro", provider: "openai" }), + ).rejects.toThrow(/\(provider=openai, model_id=deepseek-v4-pro\)/); + }); + + it("passes model timeout from system_config to GenerativeModel", async () => { + const agent = await createAgent(); + agent.state.systemConfig.model = { + ...(agent.state.systemConfig.model ?? {}), + timeoutMs: 3456, + }; + const ws = path.join(tmpRoot, "ws-timeout"); + await fs.mkdir(ws, { recursive: true }); + + const session = await agent.createSession({ workspaceDir: ws }); + const llm = (session as unknown as { engine: { deps: { llm: unknown } } }).engine.deps.llm; + + expect((llm as { requestTimeoutMs?: number }).requestTimeoutMs).toBe(3456); + }); +}); + +describe("Agent.createSession model reference((provider, model_id) 成对)", () => { + it("records the pair reference in session_meta (default_model when unspecified)", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws-ref-default"); + await fs.mkdir(ws, { recursive: true }); + const session = await agent.createSession({ workspaceDir: ws }); + try { + // Defaults to the default_model reference; session_meta carries the pair reference + // (same source that Trace writes). + const meta = session.metaMessage.payload as { provider: string; model_id: string }; + expect(meta.provider).toBe("deepseek"); + expect(meta.model_id).toBe("deepseek-v4-pro"); + expect(session.provider).toBe("deepseek"); + expect(session.modelId).toBe("deepseek-v4-pro"); + } finally { + session.dispose(); + } + }); + + it("resolves a unique bare model_id and accepts an explicit pair", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws-ref-pair"); + await fs.mkdir(ws, { recursive: true }); + // Provider omitted: model_id is a globally unique exact match in the config -> resolves to that entry. + const bare = await agent.createSession({ workspaceDir: ws, modelId: "deepseek-v4-flash" }); + try { + expect(bare.provider).toBe("deepseek"); + expect(bare.modelId).toBe("deepseek-v4-flash"); + } finally { + bare.dispose(); + } + const paired = await agent.createSession({ + workspaceDir: ws, + modelId: "claude-sonnet-4-6", + provider: "anthropic", + }); + try { + expect(paired.provider).toBe("anthropic"); + expect(paired.modelId).toBe("claude-sonnet-4-6"); + } finally { + paired.dispose(); + } + }); + + it("rejects an ambiguous bare model_id and a provider without modelId", async () => { + // Two providers coexist with the same model_id: omitting provider throws an ambiguity + // error (listing the candidate pair references). + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "myproxy", + model_id: "claude-sonnet-4-6", + }); + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws-ref-ambiguous"); + await fs.mkdir(ws, { recursive: true }); + await expect( + agent.createSession({ workspaceDir: ws, modelId: "claude-sonnet-4-6" }), + ).rejects.toThrow(/歧义.*\(provider=anthropic, model_id=claude-sonnet-4-6\)/); + // Adding provider resolves it. + const session = await agent.createSession({ + workspaceDir: ws, + modelId: "claude-sonnet-4-6", + provider: "myproxy", + }); + try { + expect(session.provider).toBe("myproxy"); + } finally { + session.dispose(); + } + // provider cannot be used alone (the reference must be a pair). + await expect(agent.createSession({ workspaceDir: ws, provider: "anthropic" })).rejects.toThrow( + /provider 不能单独使用/, + ); + }); +}); + +describe("Agent.createSession vault injection", () => { + it("injects vault key names (never values) into the assembled system prompt", async () => { + // Write the Agent vault to disk first; createSession reads that Agent's own .vault.toml. + await setVaultEntry( + tmpRoot, + DEFAULT_PROJECT_ID, + DEFAULT_AGENT_ID, + "VAULT_ONLY_KEY", + "vault-secret-value", + ); + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws-vault"); + await fs.mkdir(ws, { recursive: true }); + + const session = await agent.createSession({ workspaceDir: ws }); + try { + const meta = session.metaMessage.payload as { system_prompt: string }; + // The "# Vault" statement is part of the template body; key names are injected at the + // placeholder as a `- KEY` list. + expect(meta.system_prompt).toContain("# Vault"); + expect(meta.system_prompt).toContain("- VAULT_ONLY_KEY"); + // Values never enter the model context. + expect(meta.system_prompt).not.toContain("vault-secret-value"); + } finally { + session.dispose(); + } + }); + + it("keeps the vault statement but lists no keys when the Agent has no vault", async () => { + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws-no-vault"); + await fs.mkdir(ws, { recursive: true }); + + const session = await agent.createSession({ workspaceDir: ws }); + try { + const meta = session.metaMessage.payload as { system_prompt: string }; + // No vault: the "# Vault" section statement is kept, and the + // placeholder is replaced with an empty string, leaving no residue. + expect(meta.system_prompt).toContain("# Vault"); + expect(meta.system_prompt).not.toContain("{{VAULT_KEYS}}"); + expect(meta.system_prompt).not.toContain("VAULT_ONLY_KEY"); + } finally { + session.dispose(); + } + }); +}); + +describe("Agent.createSession skill metadata injection", () => { + it("injects installed skill metadata lines (never bodies) into the assembled system prompt", async () => { + await installSkill(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, { + name: "demo-skill", + content: + "---\nname: demo-skill\ndescription: Demo skill for tests.\nversion: 1\nupdated: 2026-07-16\n---\n\nSKILL_BODY_NOT_IN_PROMPT\n", + }); + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws-skills"); + await fs.mkdir(ws, { recursive: true }); + + const session = await agent.createSession({ workspaceDir: ws }); + try { + const meta = session.metaMessage.payload as { system_prompt: string }; + expect(meta.system_prompt).toContain("# Skills"); + expect(meta.system_prompt).toContain("- `demo-skill` — Demo skill for tests."); + // Only metadata is injected; the model reads the body on demand. + expect(meta.system_prompt).not.toContain("SKILL_BODY_NOT_IN_PROMPT"); + expect(meta.system_prompt).not.toContain("{{SKILL_METADATA}}"); + } finally { + session.dispose(); + } + }); +}); diff --git a/packages/core/test/background-registry.test.ts b/packages/core/test/background-registry.test.ts new file mode 100644 index 0000000..1396c17 --- /dev/null +++ b/packages/core/test/background-registry.test.ts @@ -0,0 +1,64 @@ +/** + * Behavior tests for BackgroundRegistry's idle reaping (a leak safety net). + */ +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { BackgroundRegistry } from "../src/environment/tools/background/index.js"; +import type { BackgroundTask } from "../src/environment/tools/background/index.js"; + +type FakeTask = BackgroundTask & { killed: boolean }; + +function fakeTask(): FakeTask { + const task: FakeTask = { + lastUsed: 0, + running: true, + killed: false, + kill() { + task.killed = true; + }, + killHard() { + task.killed = true; + }, + }; + return task; +} + +describe("BackgroundRegistry idle reaping", () => { + beforeEach(() => { + vi.useFakeTimers(); + }); + afterEach(() => { + vi.useRealTimers(); + }); + + it("reaps sessions idle past the TTL and keeps recently accessed ones", () => { + const registry = new BackgroundRegistry({ idPrefix: "proc", maxTasks: 4 }); + const stale = fakeTask(); + const fresh = fakeTask(); + const staleId = registry.register(stale); + const freshId = registry.register(fresh); + + // After 9 days, one access to fresh refreshes its lastUsed; stale is never accessed. + vi.advanceTimersByTime(9 * 24 * 60 * 60_000); + expect(registry.get(freshId)).toBe(fresh); + + // stale, now idle a full 10 days, is reaped by the scheduled sweep and finalized; + // fresh has been idle only 1 day and is kept. + vi.advanceTimersByTime(24 * 60 * 60_000 + 60 * 60_000); + expect(registry.get(staleId)).toBeUndefined(); + expect(stale.killed).toBe(true); + expect(registry.get(freshId)).toBe(fresh); + expect(fresh.killed).toBe(false); + + registry.dispose(); + }); + + it("stops the reap timer on dispose", () => { + const registry = new BackgroundRegistry({ idPrefix: "proc", maxTasks: 4 }); + const task = fakeTask(); + registry.register(task); + registry.dispose(); + expect(task.killed).toBe(true); + // After dispose the sweep timer is cleared, so fast-forwarding no longer triggers any reaping logic. + expect(() => vi.advanceTimersByTime(30 * 24 * 60 * 60_000)).not.toThrow(); + }); +}); diff --git a/packages/core/test/builtin-agents.test.ts b/packages/core/test/builtin-agents.test.ts new file mode 100644 index 0000000..d5c0f82 --- /dev/null +++ b/packages/core/test/builtin-agents.test.ts @@ -0,0 +1,141 @@ +/** + * Built-in agent provisioning and skill library install policy: the sole built-in agent + * default_agent comes pre-installed with every skill in the library, an ordinary newly created + * agent starts with zero skills, and the default AGENTS.md is an empty file; provisionProjectAgents + * is idempotent and never overwrites existing config. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { librarySkill, loadLibrarySkills } from "@prismshadow/penguin-skills"; +import { + agentsMdPath, + assembleSystemPrompt, + BUILTIN_AGENT_IDS, + DEFAULT_AGENT_ID, + DEFAULT_PROJECT_ID, + listInstalledSkills, + loadOrInitAgentState, + provisionProjectAgents, + skillsDir, +} from "../src/state/index.js"; + +let tmpRoot: string; +let prevHome: string | undefined; + +beforeEach(async () => { + prevHome = process.env.PENGUIN_HOME; + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-builtin-")); + process.env.PENGUIN_HOME = tmpRoot; +}); + +afterEach(async () => { + if (prevHome === undefined) { + delete process.env.PENGUIN_HOME; + } else { + process.env.PENGUIN_HOME = prevHome; + } + await fs.rm(tmpRoot, { recursive: true, force: true }); +}); + +const skillMdPath = (agentId: string, skillName: string): string => + path.join(skillsDir(tmpRoot, DEFAULT_PROJECT_ID, agentId), skillName, "SKILL.md"); + +describe("Skill 安装策略", () => { + it("普通新建 Agent 不预装任何 Skill,AGENTS.md 为空文件(指导在模板 Suggested workflows)", async () => { + const state = await loadOrInitAgentState({ agentId: "some_agent" }); + expect(await listInstalledSkills(tmpRoot, DEFAULT_PROJECT_ID, "some_agent")).toEqual([]); + + // The default AGENTS.md is empty: it carries no preset guidance (delegation and task + // conventions live in the default template's Suggested workflows section, and skill + // metadata is injected via {{SKILL_METADATA}}). + expect(state.agentsMd).toBe(""); + const onDisk = await fs.readFile( + agentsMdPath(tmpRoot, DEFAULT_PROJECT_ID, "some_agent"), + "utf8", + ); + expect(onDisk).toBe(""); + }); + + it("无 preset 直建的 default_agent(如 CLI 首次运行)同样预装库内全部 Skill", async () => { + await loadOrInitAgentState({ agentId: DEFAULT_AGENT_ID }); + const names = (await listInstalledSkills(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID)).map( + (s) => s.name, + ); + expect(names).toEqual(loadLibrarySkills().map((s) => s.name)); + }); +}); + +describe("provisionProjectAgents", () => { + it("唯一内置 Agent default_agent:装库内全部 Skill、AGENTS.md 为空", async () => { + const ids = await provisionProjectAgents(); + expect(ids).toEqual([DEFAULT_AGENT_ID]); + expect(BUILTIN_AGENT_IDS).toEqual([DEFAULT_AGENT_ID]); + + // name/description are written into system_config; AGENTS.md is an empty file. + const state = await loadOrInitAgentState({ agentId: DEFAULT_AGENT_ID }); + expect(state.systemConfig.name).toBe("General Agent"); + expect(state.systemConfig.description).toBeTruthy(); + expect(state.agentsMd).toBe(""); + + const installed = await listInstalledSkills(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID); + expect(installed.map((s) => s.name).sort()).toEqual(loadLibrarySkills().map((s) => s.name)); + // On-disk content matches the library's SKILL.md verbatim (install copies the full text). + const sdkMd = await fs.readFile(skillMdPath(DEFAULT_AGENT_ID, "penguin-sdk"), "utf8"); + expect(sdkMd).toBe(librarySkill("penguin-sdk")!.content); + }); + + it("provision 幂等:重复执行不改变结果", async () => { + await provisionProjectAgents(); + const ids = await provisionProjectAgents(); + expect(ids).toEqual([DEFAULT_AGENT_ID]); + const md = await fs.readFile( + agentsMdPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + "utf8", + ); + expect(md).toBe(""); + const installed = await listInstalledSkills(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID); + expect(installed.map((s) => s.name).sort()).toEqual(loadLibrarySkills().map((s) => s.name)); + }); + + it("已存在的 Agent 不被覆盖(preset 仅初始化生效)", async () => { + const custom = "# AGENTS.md\n\n用户自己改过的内容\n"; + await loadOrInitAgentState({ agentId: DEFAULT_AGENT_ID }); + await fs.writeFile(agentsMdPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), custom, "utf8"); + + await provisionProjectAgents(); + + const after = await fs.readFile( + agentsMdPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + "utf8", + ); + expect(after).toBe(custom); + }); +}); + +describe("Project Dir / Agent ID 占位符", () => { + it("assembleSystemPrompt 注入 Project Dir 与 Agent ID(Skill 定位改走项目相对路径,不依赖 .penguin)", async () => { + const state = await loadOrInitAgentState({ agentId: "env_agent" }); + const prompt = assembleSystemPrompt(state, { + sessionId: "session-x", + cwd: "/tmp/ws", + agentId: "env_agent", + projectDir: "/tmp/proj", + platform: "linux", + osVersion: "test", + date: "2026-07-08", + }); + expect(prompt).toContain("Agent ID: env_agent"); + expect(prompt).toContain("Project Dir: /tmp/proj"); + expect(prompt).not.toContain("{{AGENT_ID}}"); + expect(prompt).not.toContain("{{PROJECT_DIR}}"); + // Skill execution conventions are built from project-relative paths (no longer reference .penguin). + expect(prompt).not.toContain(".penguin"); + // Agent State, scratchpad and Skills are addressed under the Project's agents/ container. + expect(prompt).toContain("/agents//agent_state/"); + expect(prompt).toContain("/agents//agent_state/skills/"); + // The pre-agents/ layout (an agent directly under the Project dir) must never be handed to the model. + expect(prompt).not.toContain("//"); + }); +}); diff --git a/packages/core/test/compaction.test.ts b/packages/core/test/compaction.test.ts new file mode 100644 index 0000000..0aa0eed --- /dev/null +++ b/packages/core/test/compaction.test.ts @@ -0,0 +1,683 @@ +/** + * Context compaction tests. + * + * - Trigger: context usage (the request.total of the most recent token_usage) or the session's + * cumulative turn count **reaching** the threshold (>=); the check runs after every LLM + * request emits token_usage, both mid-task and at the wrap-up round (reaching the threshold at + * task end triggers compaction immediately, without waiting for the next task). + * - summarize: appends a compaction prompt to the old LLM (merging in all of this round's tool + * results first if mid-task); the summary is wrapped as a `` user text and fed + * as the first input to the new LLM instance; on failure the original context is kept, never downgraded to discard. + * - discard: deferred until task end if mid-task; sends no compaction request, just swaps in a new LLM instance directly. + * - Process visibility: the compaction request's streamed output is never surfaced to the human, + * only the paired compaction events are emitted; the dialogue is written to the old trace, and + * on success the trace rotates into a new file (index+1, the new file starts with session_meta; + * rotation is deferred until the new context has its first message to write). + */ +import { mkdtemp, rm, access, readdir } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { dirname, join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { + assistantText, + sessionMeta, + tokenUsage, + toolCall, + toolCallOutput, + userText, +} from "../src/omnimessage/index.js"; +import type { + CompactionBeginPayload, + CompactionEndPayload, + OmniMessage, + TextPayload, + TokenCounts, + TokenUsagePayload, +} from "../src/omnimessage/index.js"; +import type { + ApproveFn, + EnvironmentInterface, + GenerativeModelParameters, + LLMInterface, + LLMOutcome, +} from "../src/interfaces.js"; +import { ContextEngine } from "../src/engine/context-engine.js"; +import type { CompactionSettings } from "../src/engine/context-engine.js"; +import { Writer, readTrace } from "../src/trace/index.js"; + +// --------------------------------------------------------------------------- +// Test fixtures +// --------------------------------------------------------------------------- + +interface ScriptedResponse { + messages: OmniMessage[]; + outcome?: LLMOutcome; +} + +/** Fake LLM that responds according to a script, recording each input it receives. */ +class ScriptedLLM implements LLMInterface { + calls: OmniMessage[][] = []; + constructor( + private readonly responses: ScriptedResponse[], + readonly label = "llm", + ) {} + + async *streamGenerate( + params: GenerativeModelParameters, + ): AsyncGenerator { + this.calls.push(params.newMessages); + const next = this.responses.shift(); + if (!next) { + return { status: "failed", message: `${this.label}: no scripted response` }; + } + for (const msg of next.messages) yield msg; + return next.outcome ?? { status: "completed" }; + } +} + +/** Fake Environment that never runs real commands: any tool call returns a fixed output. */ +const fakeEnvironment: EnvironmentInterface = { + async listTools() { + return []; + }, + async *executeTool({ toolCall: tc }) { + yield toolCallOutput({ + output: "tool ran", + toolCallId: tc.payload.tool_call_id, + }); + }, + toolPermission() { + return "rw"; + }, +}; + +const allowAll: ApproveFn = async () => "allow"; + +/** Builds a token_usage: request.total is the context-usage figure, session.total is the cumulative one. */ +const usage = (requestTotal: number, sessionTotal: number): OmniMessage => + tokenUsage( + { cache_read: 0, cache_write: 0, output: 0, total: sessionTotal }, + { cache_read: 0, cache_write: 0, output: 0, total: requestTotal }, + ); + +const settings = (over: Partial = {}): CompactionSettings => ({ + maxContextLength: 100, + maxSessionTurns: -1, + mode: "summarize", + prompt: "COMPACT NOW", + ...over, +}); + +const metaMessage = sessionMeta({ + session_id: "sess_compact", + provider: "custom", + model_id: "test-model", + model_context_window: 200000, + system_prompt: "sp", + tools: [], + thinking_level: "default", + agent_state: "/tmp/state", + workspace: "/tmp/ws", +}); + +async function collect(gen: AsyncGenerator): Promise { + const all: OmniMessage[] = []; + for await (const msg of gen) all.push(msg); + return all; +} + +type CompactionEventPayload = CompactionBeginPayload | CompactionEndPayload; + +const compactionEvents = (msgs: OmniMessage[]): CompactionEventPayload[] => + msgs + .filter((m) => { + const t = (m.payload as { type?: string }).type ?? ""; + return t === "compaction_begin" || t === "compaction_end"; + }) + .map((m) => m.payload as CompactionEventPayload); + +const payloadTypes = (msgs: OmniMessage[]): (string | undefined)[] => + msgs.map((m) => (m.payload as { type?: string }).type); + +const textOf = (m: OmniMessage): string => (m.payload as TextPayload).text; + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +describe("context compaction", () => { + let traces: string; + + beforeEach(async () => { + traces = await mkdtemp(join(tmpdir(), "penguin-compaction-")); + }); + + afterEach(async () => { + await rm(traces, { recursive: true, force: true }); + }); + + it("summarize at task boundary: paired events, hidden dialogue, trace rotation, summary joins next prompt", async () => { + const llm1 = new ScriptedLLM( + [ + // Task 1's final reply: context usage 150 > threshold 100 -> triggers at the boundary. + { messages: [assistantText("answer one"), usage(150, 150)] }, + // Compaction request: summary + usage (counted into the session cumulative total). + { + messages: [assistantText("the distilled summary"), usage(160, 310)], + }, + ], + "llm1", + ); + const llm2 = new ScriptedLLM( + [{ messages: [assistantText("answer two"), usage(20, 330)] }], + "llm2", + ); + let factoryTokens: TokenCounts | null = null; + const trace = new Writer({ tracesDir: traces, sessionId: "sess_compact" }); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + trace, + sessionMeta: metaMessage, + compaction: settings(), + createLLM: (tokens) => { + factoryTokens = tokens; + return llm2; + }, + }); + const oldPath = trace.currentPath(); + + const out1 = await collect(engine.run([userText("task one")], { approve: allowAll })); + + // Paired compaction events: start carries reason/mode/context/turns, stop carries status. + const events = compactionEvents(out1); + expect(events).toHaveLength(2); + expect(events[0]).toMatchObject({ + type: "compaction_begin", + reason: "context", + mode: "summarize", + context: 150, + turns: 1, + }); + expect(events[1]).toMatchObject({ type: "compaction_end", status: "completed" }); + // The compaction process is invisible to the human: the compaction prompt and summary text are never pushed to the output stream. + const texts = out1.filter((m) => (m.payload as { type?: string }).type === "text").map(textOf); + expect(texts.some((t) => t.includes("COMPACT NOW"))).toBe(false); + expect(texts.some((t) => t.includes("distilled"))).toBe(false); + // Exception: the compaction request's token_usage IS pushed to the output stream, sitting between the paired compaction events (the frontend counts it in stats). + const types1 = payloadTypes(out1); + const between = out1.slice( + types1.indexOf("compaction_begin") + 1, + types1.lastIndexOf("compaction_end"), + ); + const usageBetween = between.filter( + (m) => (m.payload as { type?: string }).type === "token_usage", + ); + expect(usageBetween).toHaveLength(1); + expect((usageBetween[0]!.payload as TokenUsagePayload).request.total).toBe(160); + + // The new LLM instance carries over the session's cumulative tokens (including compaction request usage). + expect(factoryTokens).toMatchObject({ total: 310 }); + + // The summary is merged with the next user prompt as the new LLM instance's first input. + await collect(engine.run([userText("task two")], { approve: allowAll })); + expect(llm1.calls).toHaveLength(2); + expect(llm2.calls).toHaveLength(1); + const firstInput = llm2.calls[0]!.map(textOf); + expect(firstInput[0]).toBe("\nthe distilled summary\n"); + expect(firstInput[1]).toBe("task two"); + + // Trace splits into files: the old file contains the compaction dialogue and paired events; the new file starts with session_meta. + const oldTrace = await readTrace(oldPath); + const oldTypes = payloadTypes(oldTrace); + expect(oldTypes.filter((t) => t?.startsWith("compaction_"))).toHaveLength(2); + expect(oldTrace.some((m) => (m.payload as { text?: string }).text === "COMPACT NOW")).toBe( + true, + ); + const newTrace = await readTrace(trace.currentPath()); + expect(trace.currentPath()).not.toBe(oldPath); + expect(newTrace[0]!.type).toBe("session_meta"); + expect( + newTrace.some((m) => + ((m.payload as { text?: string }).text ?? "").startsWith(""), + ), + ).toBe(true); + }); + + it("summarize mid-task: tool outputs pair into the compaction request, summary alone feeds the new LLM", async () => { + const llm1 = new ScriptedLLM( + [ + // Round 1: tool call + over-threshold usage -> triggers mid-task. + { + messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(150, 150)], + }, + // Compaction request (should include c1's tool_call_output plus the compaction prompt). + { messages: [assistantText("continue: finish step 2")] }, + ], + "llm1", + ); + const llm2 = new ScriptedLLM( + [{ messages: [assistantText("task done"), usage(30, 200)] }], + "llm2", + ); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + sessionMeta: metaMessage, + compaction: settings(), + createLLM: () => llm2, + }); + + const out = await collect(engine.run([userText("do task")], { approve: allowAll })); + + // Compaction request: all of this round's tool results, paired with their tool_calls, are sent to the old instance along with the compaction prompt. + expect(llm1.calls).toHaveLength(2); + const compactionInput = llm1.calls[1]!; + const inputTypes = payloadTypes(compactionInput); + expect(inputTypes).toEqual(["tool_call_output", "text"]); + expect((compactionInput[1]!.payload as TextPayload).text).toBe("COMPACT NOW"); + + // The summary itself is the new instance's first input (no hardcoded continuation instruction appended); the task is finished by the new context. + expect(llm2.calls).toHaveLength(1); + expect(llm2.calls[0]!.map(textOf)).toEqual([ + "\ncontinue: finish step 2\n", + ]); + const finalTexts = out + .filter((m) => (m.payload as { type?: string }).type === "text") + .map(textOf); + expect(finalTexts).toContain("task done"); + expect(compactionEvents(out).map((e) => e.type)).toEqual([ + "compaction_begin", + "compaction_end", + ]); + }); + + it("summarize failure keeps the old context and does NOT downgrade to discard", async () => { + const llm1 = new ScriptedLLM( + [ + { + messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(150, 150)], + }, + // Compaction request fails (not retryable). + { messages: [], outcome: { status: "failed", message: "auth error" } }, + // Original context is kept: the task continues, tool outputs feed back into the old instance as usual (context usage keeps growing). + { messages: [assistantText("finished on old context"), usage(190, 340)] }, + // Second trigger (context still over the limit) -> retries compaction at the boundary, this time succeeding. + { messages: [assistantText("second try")] }, + ], + "llm1", + ); + const llm2 = new ScriptedLLM([], "llm2"); + let created = 0; + const trace = new Writer({ tracesDir: traces, sessionId: "sess_keep" }); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + trace, + sessionMeta: metaMessage, + compaction: settings(), + createLLM: () => { + created += 1; + return llm2; + }, + }); + const oldPath = trace.currentPath(); + + const out = await collect(engine.run([userText("go")], { approve: allowAll })); + + // Two event pairs: the first has stop=failed (abandoned, original context kept), the second succeeds. + const events = compactionEvents(out); + expect( + events.map((e) => `${e.type}:${(e as Partial).status ?? ""}`), + ).toEqual([ + "compaction_begin:", + "compaction_end:failed", + "compaction_begin:", + "compaction_end:completed", + ]); + // No LLM swap and no trace file split at the moment of failure; rotation happens only after success. + expect(created).toBe(1); + expect(llm1.calls).toHaveLength(4); + // After the failure, the input fed back into the old instance is this round's tool output. + expect(payloadTypes(llm1.calls[2]!)).toEqual(["tool_call_output"]); + const oldTrace = await readTrace(oldPath); + // Still written to the same file after a failed stop (the failed compaction attempt stays auditable); the old file is closed off only after success. + expect(payloadTypes(oldTrace).filter((t) => t?.startsWith("compaction_"))).toHaveLength(4); + // Trace rotation is deferred until the next message to write: the current path is unchanged right after a successful compaction. + expect(trace.currentPath()).toBe(oldPath); + }); + + it("defers trace rotation after boundary compaction until the next run writes", async () => { + const llm1 = new ScriptedLLM( + [ + { messages: [assistantText("answer"), usage(150, 150)] }, + { messages: [assistantText("s")] }, + ], + "llm1", + ); + const llm2 = new ScriptedLLM( + [{ messages: [assistantText("next done"), usage(10, 160)] }], + "llm2", + ); + const trace = new Writer({ tracesDir: traces, sessionId: "sess_lazy" }); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + trace, + sessionMeta: metaMessage, + compaction: settings(), + createLLM: () => llm2, + }); + const oldPath = trace.currentPath(); + + await collect(engine.run([userText("task 1")], { approve: allowAll })); + // A new file is not created right after boundary compaction completes: the current path is unchanged, and only the old file exists on disk. + expect(trace.currentPath()).toBe(oldPath); + expect(await readdir(dirname(oldPath))).toEqual(["sess_lazy_001.jsonl"]); + + await collect(engine.run([userText("task 2")], { approve: allowAll })); + // Rotation happens only once the next round has a message to write: the new file opens with session_meta, followed by the summary and the new prompt. + expect(trace.currentPath()).not.toBe(oldPath); + const newTrace = await readTrace(trace.currentPath()); + expect(newTrace[0]!.type).toBe("session_meta"); + expect( + ((newTrace[1]!.payload as { text?: string }).text ?? "").startsWith(""), + ).toBe(true); + expect((newTrace[2]!.payload as { text?: string }).text).toBe("task 2"); + }); + + it("reconnect exhaustion on the compaction request converges to failed", async () => { + const llm1 = new ScriptedLLM( + [ + { messages: [assistantText("answer"), usage(150, 150)] }, + { messages: [], outcome: { status: "timeout" } }, + { messages: [], outcome: { status: "timeout" } }, + ], + "llm1", + ); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => llm1, + maxReconnects: 1, + reconnectBackoffMs: 1, + }); + + const out = await collect(engine.run([userText("go")], { approve: allowAll })); + const events = compactionEvents(out); + expect(events[1]).toMatchObject({ type: "compaction_end", status: "failed" }); + // The retry resends the original input (tool results + prompt; here there are no tool results, just the prompt). + expect(llm1.calls).toHaveLength(3); + expect(payloadTypes(llm1.calls[2]!)).toEqual(["text"]); + }); + + it("session turns reaching (==) the threshold compact at task end — no waiting for the next task", async () => { + const llm1 = new ScriptedLLM( + [ + // Task 1: two LLM requests (a tool round + the final reply). + // After round 1, turns=1 < 2 doesn't trigger; round 2 (task wrap-up), turns=2 >= 2 -> compacts immediately. + { + messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(10, 10)], + }, + { messages: [assistantText("t1 done"), usage(10, 20)] }, + // Compaction request (sent out immediately when task 1 ends). + { messages: [assistantText("s")] }, + ], + "llm1", + ); + const llm2 = new ScriptedLLM([{ messages: [assistantText("t2 done"), usage(10, 30)] }], "llm2"); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings({ maxContextLength: -1, maxSessionTurns: 2 }), + createLLM: () => llm2, + }); + + // Triggers right at task 1's wrap-up (compacts as soon as the threshold is reached, without waiting for the next task); the summary request goes to the old instance. + const out1 = await collect(engine.run([userText("task 1")], { approve: allowAll })); + const events = compactionEvents(out1); + expect(events[0]).toMatchObject({ type: "compaction_begin", reason: "turns", turns: 2 }); + expect(events[1]).toMatchObject({ type: "compaction_end", status: "completed" }); + expect(llm1.calls).toHaveLength(3); + + // The counter resets after compaction completes: task 2 is picked up by the new instance (summary + new prompt), and no further trigger fires. + const out2 = await collect(engine.run([userText("task 2")], { approve: allowAll })); + expect(compactionEvents(out2)).toHaveLength(0); + expect(llm2.calls).toHaveLength(1); + expect(llm2.calls[0]!.map(textOf)).toEqual([ + "\ns\n", + "task 2", + ]); + }); + + it("context usage exactly equal to the threshold triggers compaction (>=, not >)", async () => { + const llm1 = new ScriptedLLM( + [ + // Wrap-up round context usage 100 == threshold 100 -> triggers. + { messages: [assistantText("answer"), usage(100, 100)] }, + { messages: [assistantText("eq")] }, + ], + "llm1", + ); + const llm2 = new ScriptedLLM([], "llm2"); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => llm2, + }); + + const out = await collect(engine.run([userText("go")], { approve: allowAll })); + const events = compactionEvents(out); + expect(events[0]).toMatchObject({ + type: "compaction_begin", + reason: "context", + context: 100, + }); + expect(events[1]).toMatchObject({ type: "compaction_end", status: "completed" }); + }); + + it("discard defers mid-task, then swaps the LLM at task end without a compaction request", async () => { + const llm1 = new ScriptedLLM( + [ + // Round 1: over the limit, but the task is still in progress -> deferred. + { + messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(150, 150)], + }, + // Round 2: task ends -> performs discard (no compaction request sent). + { messages: [assistantText("done"), usage(160, 310)] }, + ], + "llm1", + ); + const llm2 = new ScriptedLLM([{ messages: [assistantText("fresh"), usage(10, 320)] }], "llm2"); + const trace = new Writer({ tracesDir: traces, sessionId: "sess_discard" }); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + trace, + sessionMeta: metaMessage, + compaction: settings({ mode: "discard" }), + createLLM: () => llm2, + }); + const oldPath = trace.currentPath(); + + const out = await collect(engine.run([userText("go")], { approve: allowAll })); + + // Triggers exactly once, at task end; the old LLM is called exactly twice (no compaction request). + const events = compactionEvents(out); + expect(events.map((e) => `${e.type}:${e.mode}`)).toEqual([ + "compaction_begin:discard", + "compaction_end:discard", + ]); + expect(llm1.calls).toHaveLength(2); + + // The next round's input is used as-is as the new instance's first input (no ). + await collect(engine.run([userText("next task")], { approve: allowAll })); + expect(llm2.calls[0]!.map(textOf)).toEqual(["next task"]); + + // Trace splits into files: the new file starts with session_meta. + const newTrace = await readTrace(trace.currentPath()); + expect(trace.currentPath()).not.toBe(oldPath); + expect(newTrace[0]!.type).toBe("session_meta"); + }); + + it("lenient extraction: output without tags is used verbatim", async () => { + const llm1 = new ScriptedLLM( + [ + { messages: [assistantText("answer"), usage(150, 150)] }, + { messages: [assistantText("plain summary text without tags")] }, + ], + "llm1", + ); + const llm2 = new ScriptedLLM([{ messages: [assistantText("ok"), usage(10, 160)] }], "llm2"); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => llm2, + }); + + await collect(engine.run([userText("go")], { approve: allowAll })); + await collect(engine.run([userText("next")], { approve: allowAll })); + expect(textOf(llm2.calls[0]![0]!)).toBe( + "\nplain summary text without tags\n", + ); + }); + + it("manual compaction skips threshold checks and reuses the same flow", async () => { + const llm1 = new ScriptedLLM( + [ + // One ordinary task (well under the limit). + { messages: [assistantText("small"), usage(10, 10)] }, + // Manual compaction request. + { messages: [assistantText("manual s")] }, + ], + "llm1", + ); + const llm2 = new ScriptedLLM([{ messages: [assistantText("after"), usage(5, 20)] }], "llm2"); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => llm2, + }); + + await collect(engine.run([userText("hi")], { approve: allowAll })); + const out = await collect(engine.compact()); + const events = compactionEvents(out); + expect(events[0]).toMatchObject({ type: "compaction_begin", reason: "manual" }); + expect(events[1]).toMatchObject({ type: "compaction_end", status: "completed" }); + + await collect(engine.run([userText("next")], { approve: allowAll })); + expect(llm2.calls[0]!.map(textOf)).toEqual([ + "\nmanual s\n", + "next", + ]); + }); + + it("no compaction capability (createLLM missing) means thresholds never fire and compact() is a no-op", async () => { + const llm1 = new ScriptedLLM( + [{ messages: [assistantText("big"), usage(999999, 999999)] }], + "llm1", + ); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + }); + const out = await collect(engine.run([userText("go")], { approve: allowAll })); + expect(compactionEvents(out)).toHaveLength(0); + expect(await collect(engine.compact())).toHaveLength(0); + expect(llm1.calls).toHaveLength(1); + }); + + it("compactability(): 逐一给出「不能压缩」的原因,而不是笼统的 false", async () => { + // compact() emits **zero messages** when there is nothing compactable. The caller (Web / CLI) + // must be able to ask ahead of time, otherwise it can only wait forever for a compaction + // banner that never comes -- exactly how "no response from /compact after interrupting on the + // web" happens: interrupt the first request -> token_usage is never received -> sessionTurns + // stays at 0 -> compact() returns immediately. + // The reason also needs to distinguish "just compacted" from "haven't chatted yet" -- both + // have sessionTurns == 0, but they're two completely different messages to the user: telling + // someone who just finished compacting that there's "no completed conversation turn yet" is + // effectively saying nothing useful. + // The compaction request goes to the **current** LLM (the script's second entry); createLLM + // supplies the LLM used for the new context after compaction. + const llm1 = new ScriptedLLM( + [ + // Usage is kept under maxContextLength (100 per settings()) so automatic compaction doesn't jump in first and reset sessionTurns. + { messages: [assistantText("hi"), usage(10, 10)] }, + { messages: [assistantText("s")] }, // manual compaction request + ], + "llm1", + ); + const llm2 = new ScriptedLLM([{ messages: [assistantText("after"), usage(5, 20)] }], "llm2"); + + // (1) No compaction capability configured (no compaction / createLLM). + const noCap = new ContextEngine({ llm: llm1, environment: fakeEnvironment }); + expect(noCap.compactability()).toBe("unsupported"); + + // (2) Capability configured, but the current context hasn't finished a single round yet: not compactable, and compact() indeed emits no messages. + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => llm2, + }); + expect(engine.compactability()).toBe("empty"); + expect(await collect(engine.compact())).toHaveLength(0); + + // (3) Compactable only after a round finishes (token_usage received). + await collect(engine.run([userText("go")], { approve: allowAll })); + expect(engine.compactability()).toBe("ok"); + expect(compactionEvents(await collect(engine.compact()))).not.toHaveLength(0); + + // (4) Compacting again right after a compaction: also not compactable (the new context is + // empty), but the reason is "just compacted" -- it must not say "no completed conversation + // turn yet" again, since the user clearly just finished a whole round. + expect(engine.compactability()).toBe("just_compacted"); + expect(await collect(engine.compact())).toHaveLength(0); + + // (5) Compactable again once a round finishes in the new context. + await collect(engine.run([userText("next")], { approve: allowAll })); + expect(engine.compactability()).toBe("ok"); + }); + + it("user abort during the compaction request keeps the context and carries tool outputs over", async () => { + const controller = new AbortController(); + const llm1 = new ScriptedLLM( + [ + { + messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(150, 150)], + }, + { messages: [], outcome: { status: "aborted" } }, + ], + "llm1", + ); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => llm1, + }); + // The abort signal is already pending before the compaction request: the fake LLM finishes straight to aborted. + const approveThenAbort: ApproveFn = async () => "allow"; + const runGen = engine.run([userText("go")], { + approve: approveThenAbort, + signal: controller.signal, + }); + const out: OmniMessage[] = []; + for await (const msg of runGen) { + out.push(msg); + // Simulates a user abort right after the compaction start event (the compaction request then returns aborted). + const p = msg.payload as { type?: string }; + if (p.type === "compaction_begin") controller.abort(); + } + + const events = compactionEvents(out); + expect(events[1]).toMatchObject({ type: "compaction_end", status: "aborted" }); + // Interrupt cleanup: the tool output is held as carry-over per case A, and the run wraps up with an abort event. + expect(payloadTypes(out)).toContain("abort"); + }); +}); diff --git a/packages/core/test/describe-image.test.ts b/packages/core/test/describe-image.test.ts new file mode 100644 index 0000000..26f3795 --- /dev/null +++ b/packages/core/test/describe-image.test.ts @@ -0,0 +1,241 @@ +/** + * Unit tests (offline) for the read_image "vision-model describe" variant, driven by a fake LLM: + * definition overrides (new prompt parameter, description mentioning the vision model id), a + * single image + prompt sent to the vision model, text output and failure paths (no vision model + * configured / vision model request fails), results carrying no images; and swapping on the + * Environment side (injecting visionDescriber switches to the describe variant). + */ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { + DESCRIBE_IMAGE_NAME, + createDescribeImageTool, +} from "../src/environment/tools/describe-image.js"; +import { BUILTIN_TOOL_FACTORIES } from "../src/environment/tools/registry.js"; +import { Environment } from "../src/environment/environment.js"; +import { assistantText, partialText, toolCall } from "../src/omnimessage/index.js"; +import type { OmniMessage } from "../src/omnimessage/index.js"; +import type { ToolResult } from "../src/environment/tools/types.js"; +import type { + GenerativeModelParameters, + LLMInterface, + LLMOutcome, + ToolDefinitionConfig, + VisionDescriberService, +} from "../src/interfaces.js"; + +/** 1x1 transparent PNG (includes magic bytes, enough for mime sniffing). */ +const PNG_1X1 = Buffer.from( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==", + "base64", +); + +/** Config entry for describe_image (forModel: "text-only"; the definition comes entirely from config, the implementation never rewrites it at runtime). */ +const definition: ToolDefinitionConfig = { + name: DESCRIBE_IMAGE_NAME, + forModel: "text-only", + description: "describe image via vision model", + parameters: { + type: "object", + properties: { source: { type: "string" }, prompt: { type: "string" } }, + required: ["source"], + }, + permission: "r", +}; + +/** + * Fake vision LLM: records the received newMessages, emits output following the real streaming + * protocol (partial start -> word-by-word delta -> stop -> complete text), and finishes with the given outcome. + */ +function fakeLLM(reply: string, outcome: LLMOutcome = { status: "completed" }) { + const calls: GenerativeModelParameters[] = []; + const llm: LLMInterface = { + // eslint-disable-next-line @typescript-eslint/require-await + async *streamGenerate(params: GenerativeModelParameters) { + calls.push(params); + if (reply) { + yield partialText("start"); + // Split into two delta chunks to verify piecewise forwarding (rather than buffering the whole thing). + const mid = Math.ceil(reply.length / 2); + yield partialText("delta", reply.slice(0, mid)); + yield partialText("delta", reply.slice(mid)); + yield partialText("stop"); + yield assistantText(reply); + } + return outcome; + }, + }; + return { llm, calls }; +} + +async function run( + args: Record, + workspaceDir: string, + describer: VisionDescriberService, +) { + const tool = createDescribeImageTool(definition, describer); + const gen = tool.execute(args, { workspaceDir, toolCallId: "c1" }); + const messages: OmniMessage[] = []; + let result: ToolResult | void; + for (;;) { + const res = await gen.next(); + if (res.done) { + result = res.value; + break; + } + messages.push(res.value); + } + const text = messages.map((m) => (m.payload as { output?: string }).output ?? "").join(""); + return { messages, result, text }; +} + +let tmp: string; + +beforeEach(async () => { + tmp = await mkdtemp(path.join(tmpdir(), "penguin-descimg-")); +}); + +afterEach(async () => { + await rm(tmp, { recursive: true, force: true }); +}); + +describe("describe_image(read_image 的纯文本模型版)", () => { + it("定义原样取自配置条目(不做运行期改写)", () => { + const tool = createDescribeImageTool(definition, { modelId: "vis-1" }); + expect(tool.name).toBe(DESCRIBE_IMAGE_NAME); + expect(tool.definition).toBe(definition); + }); + + it("registry 按工具名装配;未注入 visionDescriber 时以 failed 说明收尾(而非回图片)", async () => { + const factory = BUILTIN_TOOL_FACTORIES[DESCRIBE_IMAGE_NAME]!; + await writeFile(path.join(tmp, "a.png"), PNG_1X1); + const describeTool = factory(definition, undefined); + const gen = describeTool.execute({ source: "a.png" }, { workspaceDir: tmp, toolCallId: "c1" }); + let result: ToolResult | void; + let text = ""; + for (;;) { + const res = await gen.next(); + if (res.done) { + result = res.value; + break; + } + text += (res.value.payload as { output?: string }).output ?? ""; + } + expect(result?.stopReason).toBe("failed"); + expect(text).toContain("No vision model"); + }); + + it("图片 + 自定义 prompt 单发视觉模型,回其文本,结果不携带 images", async () => { + await writeFile(path.join(tmp, "a.png"), PNG_1X1); + const { llm, calls } = fakeLLM("图里是一只企鹅。"); + const describer: VisionDescriberService = { modelId: "vis-1", createLLM: () => llm }; + const { messages, result, text } = await run( + { source: "a.png", prompt: "图里是什么动物?" }, + tmp, + describer, + ); + + // Single message = prompt text + data URL image (same role user, merged into one request). + expect(calls).toHaveLength(1); + const payloads = calls[0]!.newMessages.map( + (m) => m.payload as { type: string; text?: string; image_url?: string }, + ); + expect(payloads[0]!.type).toBe("text"); + expect(payloads[0]!.text).toBe("图里是什么动物?"); + expect(payloads[1]!.type).toBe("image_url"); + expect(payloads[1]!.image_url).toBe(`data:image/png;base64,${PNG_1X1.toString("base64")}`); + + expect(text).toContain("described by vis-1"); + expect(text).toContain("图里是一只企鹅。"); + // Streaming forward: the header line and description deltas are emitted as separate chunks (not buffered as a whole), and the complete text is not forwarded again. + const outputs = messages.map((m) => (m.payload as { output?: string }).output ?? ""); + expect(outputs.length).toBeGreaterThanOrEqual(3); // header + >=2 description delta chunks + expect(outputs[0]).toContain("described by vis-1"); + expect(outputs.slice(1).join("")).toBe("图里是一只企鹅。"); + // Text-based description: the result carries no images (images never enter session history). + expect(result?.images).toBeUndefined(); + expect(result?.stopReason).toBeUndefined(); // defaults to completed + }); + + it("未给 prompt 时用默认问题", async () => { + await writeFile(path.join(tmp, "a.png"), PNG_1X1); + const { llm, calls } = fakeLLM("desc"); + await run({ source: "a.png" }, tmp, { modelId: "vis-1", createLLM: () => llm }); + const first = calls[0]!.newMessages[0]!.payload as { text?: string }; + expect(first.text).toContain("Describe this image"); + }); + + it("未配置视觉模型:failed 并解释如何配置", async () => { + await writeFile(path.join(tmp, "a.png"), PNG_1X1); + const { result, text } = await run({ source: "a.png" }, tmp, { modelId: null }); + expect(result?.stopReason).toBe("failed"); + expect(text).toContain("No vision model"); + expect(text).toContain("vision_model"); + }); + + it("视觉模型请求失败:failed 并带状态与消息", async () => { + await writeFile(path.join(tmp, "a.png"), PNG_1X1); + const { llm } = fakeLLM("", { status: "failed", message: "401 unauthorized" }); + const { result, text } = await run({ source: "a.png" }, tmp, { + modelId: "vis-1", + createLLM: () => llm, + }); + expect(result?.stopReason).toBe("failed"); + expect(text).toContain("failed"); + expect(text).toContain("401 unauthorized"); + }); + + it("图片校验复用 read_image:不支持的类型直接 failed,不请求视觉模型", async () => { + await writeFile(path.join(tmp, "a.txt"), "not an image"); + const { llm, calls } = fakeLLM("desc"); + const { result } = await run({ source: "a.txt" }, tmp, { + modelId: "vis-1", + createLLM: () => llm, + }); + expect(result?.stopReason).toBe("failed"); + expect(calls).toHaveLength(0); + }); + + it("Environment 装配 describe_image 条目走代读实现,定义与配置一致", async () => { + await writeFile(path.join(tmp, "a.png"), PNG_1X1); + const { llm } = fakeLLM("代读结果"); + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + // Already filtered by selectBuiltinToolsForModel per session model before assembly; only the describe entry remains here. + customTools: [definition], + mcpServers: [], + }, + services: { visionDescriber: { modelId: "vis-1", createLLM: () => llm } }, + }); + // Tool listing matches config: describe_image carries the prompt parameter. + const tools = await env.listTools(); + const describeImage = tools.find((t) => t.name === DESCRIBE_IMAGE_NAME)!; + const props = (describeImage.parameters as { properties: Record }).properties; + expect(Object.keys(props)).toContain("prompt"); + + // Execution goes through description: outputs text, the complete message carries no images. + const out: OmniMessage[] = []; + for await (const m of env.executeTool({ + toolCall: toolCall({ + name: DESCRIBE_IMAGE_NAME, + arguments: '{"source":"a.png"}', + toolCallId: "t1", + }), + })) { + out.push(m); + } + const complete = out[out.length - 1]!.payload as { + type?: string; + output?: string; + images?: string[]; + stop_reason?: string; + }; + expect(complete.type).toBe("tool_call_output"); + expect(complete.stop_reason).toBe("completed"); + expect(complete.output).toContain("代读结果"); + expect(complete.images).toBeUndefined(); + }); +}); diff --git a/packages/core/test/engine.test.ts b/packages/core/test/engine.test.ts new file mode 100644 index 0000000..d54407f --- /dev/null +++ b/packages/core/test/engine.test.ts @@ -0,0 +1,1683 @@ +/** + * context_engine integration tests (mock LLM, no API key needed). + * + * New protocol: a single `run(prompt, { signal, approve })` automatically drives the whole + * ReAct loop — it consumes the LLM stream, calls `approve` immediately for each tool_call, + * executes it via Environment when allowed, feeds the result back and continues to the next + * turn, until some turn produces no tool_call (Task done) or is interrupted. Approval/execution + * are within-turn interactions, and execution can overlap. + */ +import { mkdtemp, readFile, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { + assistantText, + emptyTokenCounts, + isCompleteModelMessage, + partialText, + partialToolCallOutput, + sessionMeta, + thinkingMessage, + toolCall, + toolCallOutput, + tokenUsage, + userText, + withOrigin, +} from "../src/omnimessage/index.js"; +import { BUILTIN_TOOL_FACTORIES } from "../src/environment/tools/registry.js"; +import type { GenerativeModelParameters, LLMInterface, LLMOutcome } from "../src/interfaces.js"; +import type { OmniMessage, TextPayload, ToolCallPayload } from "../src/omnimessage/index.js"; +import { Environment } from "../src/environment/index.js"; +import { Writer, readTrace } from "../src/trace/index.js"; +import { ContextEngine } from "../src/engine/context-engine.js"; +import type { ApproveFn, EnvironmentInterface, ToolPermission } from "../src/interfaces.js"; + +/** Deterministic fake LLM: the first turn yields a tool_call, the second yields the final reply. */ +class FakeLLM implements LLMInterface { + calls = 0; + receivedSecondInput: OmniMessage[] | null = null; + + async *streamGenerate( + params: GenerativeModelParameters, + ): AsyncGenerator { + this.calls += 1; + if (this.calls === 1) { + yield partialText("start", ""); + yield partialText("delta", "I will create the file."); + yield partialText("stop", "", "completed"); + yield assistantText("I will create the file."); + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "printf 'Hello, Penguin' > hello.txt" }), + toolCallId: "call_1", + stopReason: "completed", + }); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 5, + total: 12, + }); + return { status: "completed" }; + } + this.receivedSecondInput = params.newMessages; + yield assistantText("Done. Created hello.txt with the greeting."); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 8, + total: 20, + }); + return { status: "completed" }; + } +} + +function execCommandToolConfig() { + return { + customTools: [ + { + name: "exec_command", + description: "Run a shell command.", + parameters: { + type: "object", + properties: { cmd: { type: "string" }, workdir: { type: "string" } }, + required: ["cmd"], + }, + permission: "rw" as const, + maxOutputLength: 16000, + }, + ], + mcpServers: [], + }; +} + +const isToolCall = (m: OmniMessage): boolean => + isCompleteModelMessage(m) && m.payload.type === "tool_call"; + +/** Count of text messages in the list starting with `` (flatten carry-over count). */ +const turnAbortedCount = (msgs: OmniMessage[]): number => + msgs.filter((m) => ((m.payload as { text?: string }).text ?? "").startsWith("")) + .length; + +/** An approval callback that allows everything. */ +const allowAll: ApproveFn = async () => "allow"; +/** An approval callback that denies everything. */ +const denyAll: ApproveFn = async () => "deny"; + +/** Collects all output from a run. */ +async function collectRun( + engine: ContextEngine, + prompt: OmniMessage[], + approve: ApproveFn, + signal?: AbortSignal, +): Promise { + const all: OmniMessage[] = []; + for await (const msg of engine.run(prompt, { + approve, + ...(signal ? { signal } : {}), + })) { + all.push(msg); + } + return all; +} + +describe("ContextEngine ReAct loop (mock LLM, approve callback)", () => { + let workspace: string; + let traces: string; + + beforeEach(async () => { + workspace = await mkdtemp(join(tmpdir(), "penguin-ws-")); + traces = await mkdtemp(join(tmpdir(), "penguin-tr-")); + }); + + afterEach(async () => { + await rm(workspace, { recursive: true, force: true }); + await rm(traces, { recursive: true, force: true }); + }); + + it("approves a tool call, writes the file, returns the final answer, traces it", async () => { + const llm = new FakeLLM(); + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const trace = new Writer({ tracesDir: traces, sessionId: "sess_test" }); + const engine = new ContextEngine({ llm, environment, trace }); + + const collected = await collectRun( + engine, + [userText("Create hello.txt saying Hello, Penguin")], + allowAll, + ); + + expect(llm.calls).toBe(2); + expect( + llm.receivedSecondInput!.some( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + ), + ).toBe(true); + expect(await readFile(join(workspace, "hello.txt"), "utf8")).toBe("Hello, Penguin"); + + const types = collected.map((m) => (m.payload as { type?: string }).type); + expect(types).toContain("tool_call_output"); + // approve is a callback; context_engine emits the approval result as an approval_decision + // event (for frontend rendering + Trace). + expect(types).toContain("approval_decision"); + const finalTexts = collected + .filter((m) => isCompleteModelMessage(m) && m.payload.type === "text") + .map((m) => (m.payload as TextPayload).text); + expect(finalTexts.some((t) => t.includes("Done"))).toBe(true); + + const recorded = await readTrace(trace.currentPath()); + const recordedTypes = recorded.map((m) => (m.payload as { type?: string }).type); + expect(recordedTypes).toContain("tool_call"); + expect(recordedTypes).toContain("tool_call_output"); + expect(recordedTypes.some((t) => t?.startsWith("partial_"))).toBe(false); + }); + + it("streams origin-tagged nested messages to the consumer but keeps them out of trace and the next-turn input", async () => { + const NAME = "__nested_forward_tool__"; + BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({ + name: NAME, + definition, + async *execute(_args, ctx) { + // Simulates run_subagent: forwards one complete tool_call_output from a child session + // (with origin), then yields its own output. + yield withOrigin( + toolCallOutput({ output: "child result", toolCallId: "child_call" }), + "sess_child", + ); + yield partialToolCallOutput({ + eventType: "delta", + output: "own result", + toolCallId: ctx.toolCallId, + }); + }, + }); + try { + let calls = 0; + let secondInput: OmniMessage[] | null = null; + const llm: LLMInterface = { + async *streamGenerate(params): AsyncGenerator { + calls += 1; + if (calls === 1) { + yield toolCall({ + name: NAME, + arguments: "{}", + toolCallId: "p1", + stopReason: "completed", + }); + return { status: "completed" }; + } + secondInput = params.newMessages; + yield assistantText("Done"); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: { + customTools: [{ name: NAME, description: "fwd", permission: "rw" }], + mcpServers: [], + }, + }); + const trace = new Writer({ tracesDir: traces, sessionId: "sess_fwd" }); + const engine = new ContextEngine({ llm, environment, trace }); + + const all = await collectRun(engine, [userText("go")], allowAll); + + // The nested message reaches the frontend via the stream (with origin), for rendering. + const nested = all.find((m) => m.origin?.length); + expect(nested).toBeDefined(); + expect((nested!.payload as { output?: string }).output).toBe("child result"); + // The input fed back for the next turn contains only this level's tool output, not the + // child session's tool_call_output (unpaired; feeding it back by mistake would be rejected). + const secondOutputs = secondInput!.filter( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + ); + expect(secondOutputs).toHaveLength(1); + expect((secondOutputs[0]!.payload as { output?: string }).output).toBe("own result"); + // The parent Trace does not record the nested message (the child Session has its own Trace). + const recorded = await readTrace(trace.currentPath()); + expect( + recorded.some((m) => (m.payload as { output?: string }).output === "child result"), + ).toBe(false); + } finally { + delete BUILTIN_TOOL_FACTORIES[NAME]; + } + }); + + it("denied approval feeds an aborted result back to the model, no file written", async () => { + const llm = new FakeLLM(); + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment }); + + const all = await collectRun(engine, [userText("Create hello.txt")], denyAll); + + await expect(readFile(join(workspace, "hello.txt"), "utf8")).rejects.toThrow(); + const denialOutput = llm.receivedSecondInput!.find( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + ); + expect((denialOutput!.payload as { output: string }).output).toContain("denied"); + // A denial's stop_reason is "aborted", indicating the tool call was manually canceled. + const deniedMsg = all.find( + (m) => + (m.payload as { type?: string }).type === "tool_call_output" && + (m.payload as { stop_reason?: string }).stop_reason === "aborted", + ); + expect(deniedMsg).toBeDefined(); + }); + + it("max_turns default is 100", () => { + const engine = new ContextEngine({ + llm: new FakeLLM(), + environment: new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }), + }); + // Reads the default via a private field (white-box, only testing the default). + expect((engine as unknown as { maxTurns: number }).maxTurns).toBe(100); + }); + + it("streams the max-turns stop note before the complete text (no extra leading newline)", async () => { + const llm: LLMInterface = { + async *streamGenerate() { + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "true" }), + toolCallId: "c1", + stopReason: "completed", + }); + // The model output completes normally -> this turn is not interrupted, so the loop can + // advance to the next turn and trigger max_turns. + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment, maxTurns: 1 }); + + const all = await collectRun(engine, [userText("go")], allowAll); + const partials = all + .filter((m) => (m.payload as { type?: string }).type === "partial_text") + .map((m) => m.payload); + const maxTurnText = "[reached max turns (1); stopping]"; + + expect(partials).toMatchObject([ + { event_type: "start", text: "" }, + { event_type: "delta", text: maxTurnText }, + { event_type: "stop", text: "", stop_reason: "failed" }, + ]); + // No more leading newline. + expect(maxTurnText.startsWith("\n")).toBe(false); + expect( + all.some( + (m) => + isCompleteModelMessage(m) && m.payload.type === "text" && m.payload.text === maxTurnText, + ), + ).toBe(true); + }); + + it("max turns with pending tool outputs carries them over so the next run pairs the tool_call (issue #33)", async () => { + const received: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params): AsyncGenerator { + received.push(params.newMessages); + if (received.length === 1) { + // Turn 1: the tool call completes normally; the tool output cannot be fed back + // because max_turns was hit. + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "true" }), + toolCallId: "c1", + stopReason: "completed", + }); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + } + yield assistantText("continuing"); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment, maxTurns: 1 }); + + await collectRun(engine, [userText("go")], allowAll); + + // Continuing input on the same Session: the previous turn's tool output is resent, merged + // with the new input, as a structured carry-over (case A); since the committed tool_call now + // has a paired output, it does not trigger the provider's unanswered-tool_use rejection. + await collectRun(engine, [userText("continue the fix")], allowAll); + expect(received).toHaveLength(2); + const secondTypes = received[1]!.map((m) => (m.payload as { type?: string }).type); + expect(secondTypes).toEqual(["tool_call_output", "text"]); + expect((received[1]![0]!.payload as { tool_call_id?: string }).tool_call_id).toBe("c1"); + expect((received[1]![1]!.payload as TextPayload).text).toBe("continue the fix"); + }); + + it("aborts before run: emits abort and carries the (wrapped) input over to the next run", async () => { + const received: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + received.push(params.newMessages); + yield assistantText("ok"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment }); + const controller = new AbortController(); + controller.abort(); + + const all = await collectRun(engine, [userText("go")], allowAll, controller.signal); + // Interrupted before dispatch: emits abort, and the model is never actually called. + expect(all.map((m) => (m.payload as { type?: string }).type)).toContain("abort"); + expect(received).toHaveLength(0); + + // Next turn: input that never made it to a Request is kept **as-is** as carry-over, and sent + // together with the new input (trailing-input semantics; not flattened, so replay matches + // in-process behavior). + await collectRun(engine, [userText("next")], allowAll); + expect(received).toHaveLength(1); + const texts = received[0]!.map((m) => (m.payload as { text?: string }).text ?? ""); + expect(texts).toContain("go"); + expect(texts).toContain("next"); + expect(texts.join("\n")).not.toContain(""); + }); + + it("never writes the flatten carry-over to trace (case B): synthesized carry-over is memory-only", async () => { + let call = 0; + const llm: LLMInterface = { + async *streamGenerate() { + call += 1; + if (call === 1) { + // The model output is interrupted mid-stream (case B): partial thinking + aborted finish. + yield thinkingMessage("half thought", "aborted"); + return { status: "aborted" }; + } + yield assistantText("ok"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const trace = new Writer({ tracesDir: traces, sessionId: "sess_carry_tr" }); + const engine = new ContextEngine({ llm, environment, trace }); + + await collectRun(engine, [userText("first ask")], allowAll); + // Synthesized carry-over is not written to Trace (Trace only records real messages). + expect(turnAbortedCount(await readTrace(trace.currentPath()))).toBe(0); + + await collectRun(engine, [userText("next")], allowAll); + // Likewise not persisted when sent: flattening is only sent to the model. + expect(turnAbortedCount(await readTrace(trace.currentPath()))).toBe(0); + }); + + it("never writes case-A backfill placeholders to trace: pairing is re-synthesized on resume", async () => { + const controller = new AbortController(); + let call = 0; + const llm: LLMInterface = { + async *streamGenerate() { + call += 1; + if (call === 1) { + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "true" }), + toolCallId: "a1", + stopReason: "completed", + }); + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "true" }), + toolCallId: "a2", + stopReason: "completed", + }); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + } + yield assistantText("resumed"); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const trace = new Writer({ tracesDir: traces, sessionId: "sess_backfill_tr" }); + const engine = new ContextEngine({ llm, environment, trace }); + // Interrupted while approving the first tool: a1 and a2 are both committed but not + // dispatched, so the carry-over is two interrupted-state placeholders. + const approve: ApproveFn = async () => { + controller.abort(); + return "allow"; + }; + await collectRun(engine, [userText("go")], approve, controller.signal); + const placeholders = (msgs: OmniMessage[]): number => + msgs.filter( + (m) => (m.payload as { output?: string }).output === "[interrupted: tool aborted by user]", + ).length; + // The placeholder is synthesized only in memory, never written to Trace (resume/replay + // re-synthesizes it on demand as a pairing fallback). + expect(placeholders(await readTrace(trace.currentPath()))).toBe(0); + + await collectRun(engine, [userText("continue")], allowAll); + const recorded = await readTrace(trace.currentPath()); + // The backfill is sent along with the request, and is likewise never persisted. + expect(placeholders(recorded)).toBe(0); + expect( + recorded.filter((m) => (m.payload as { type?: string }).type === "tool_call_output"), + ).toHaveLength(0); + }); + + it("writes a subagent pointer event (session id only) when a direct child's session_meta arrives", async () => { + let llmCalls = 0; + const llm: LLMInterface = { + async *streamGenerate() { + llmCalls += 1; + if (llmCalls === 1) { + yield toolCall({ name: "spawn", arguments: "{}", toolCallId: "tc-spawn" }); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + } + yield assistantText("done"); + return { status: "completed" }; + }, + }; + const childMeta = (sid: string) => + sessionMeta({ + session_id: sid, + provider: "custom", + model_id: "m-child", + model_context_window: 1000, + system_prompt: "sys", + tools: [], + thinking_level: "medium", + agent_state: "/root/p/worker/agent_state", + workspace: "/tmp/w", + }); + // Custom Environment: on execution, first forwards origin-tagged child session messages + // (child meta / child text / grandchild meta), then yields the complete output (simulates + // run_subagent's forwarding behavior). + const environment: EnvironmentInterface = { + listTools: async () => [], + toolPermission: () => undefined, + async *executeTool(request) { + yield withOrigin(childMeta("sess-child"), "sess-child"); + yield withOrigin(assistantText("from child"), "sess-child"); + yield withOrigin(withOrigin(childMeta("sess-grand"), "sess-grand"), "sess-child"); + yield toolCallOutput({ + output: "spawned", + toolCallId: request.toolCall.payload.tool_call_id, + }); + }, + }; + const trace = new Writer({ tracesDir: traces, sessionId: "sess_subagent_ptr" }); + const engine = new ContextEngine({ llm, environment, trace }); + await collectRun(engine, [userText("go")], allowAll); + + const rows = await readTrace(trace.currentPath()); + // A direct child session's (origin length 1) session_meta -> exactly one subagent pointer + // event (recording only the Session id); a grandchild session's (origin length 2) does not + // get a pointer, since its own child Trace records it. + const pointers = rows.filter((m) => (m.payload as { type?: string }).type === "subagent"); + expect(pointers).toHaveLength(1); + expect(pointers[0]!.type).toBe("event_msg"); + expect((pointers[0]!.payload as { session_id?: string }).session_id).toBe("sess-child"); + // Origin-tagged child session messages (session_meta and body alike) are never written + // to the parent Trace. + expect(rows.some((m) => m.origin !== undefined)).toBe(false); + }); + + it("in-run reconnect never writes the synthesized to trace", async () => { + let calls = 0; + const inputs: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + calls += 1; + inputs.push(params.newMessages); + if (calls === 1) { + yield assistantText("half", "timeout"); + return { status: "timeout" }; + } + yield assistantText("recovered"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const trace = new Writer({ tracesDir: traces, sessionId: "sess_retry_tr" }); + const engine = new ContextEngine({ + llm, + environment, + trace, + maxReconnects: 1, + reconnectBackoffMs: 1, + }); + + await collectRun(engine, [userText("go")], allowAll); + expect(calls).toBe(2); + // Retry = original input + (carrying the partial text). + expect((inputs[1]![1]!.payload as { text?: string }).text ?? "").toContain(""); + // The synthesized message is only sent to the model: Trace has no / + // ; the original input is written only on its first occurrence. + const recorded = await readTrace(trace.currentPath()); + expect(turnAbortedCount(recorded)).toBe(0); + expect( + recorded.some((m) => + ((m.payload as { text?: string }).text ?? "").startsWith(""), + ), + ).toBe(false); + expect(recorded.filter((m) => (m.payload as { text?: string }).text === "go")).toHaveLength(1); + }); +}); + +describe("ContextEngine async/incremental tool calls (overlapping execution)", () => { + let workspace: string; + beforeEach(async () => { + workspace = await mkdtemp(join(tmpdir(), "penguin-ws2-")); + }); + afterEach(async () => { + await rm(workspace, { recursive: true, force: true }); + }); + + it("emits both tool calls in one round; second is approved while the first executes; outputs come back in completion order", async () => { + // The first turn yields two tool_calls; the second yields the final reply. + const llm: LLMInterface = { + calls: 0, + async *streamGenerate(this: { calls: number }) { + this.calls += 1; + if (this.calls === 1) { + // First tool: slow (sleep 0.4s). Second tool: fast. + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "sleep 0.4; printf one > a.txt" }), + toolCallId: "t1", + stopReason: "completed", + }); + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "printf two > b.txt" }), + toolCallId: "t2", + stopReason: "completed", + }); + // The model output completes normally -> this turn is not interrupted, results are + // fed back into the next turn. + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + } + yield assistantText("both done"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + } as LLMInterface & { calls: number }; + + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment }); + + // Record approval order and timestamps to prove the second tool enters approval while the + // first is still executing (execution can overlap). + const approvedAt: Record = {}; + const firstCompleteAt: Record = {}; + const start = Date.now(); + const approve: ApproveFn = async (tc) => { + approvedAt[tc.payload.tool_call_id] = Date.now() - start; + return "allow"; + }; + + const all: OmniMessage[] = []; + for await (const msg of engine.run([userText("go")], { approve })) { + all.push(msg); + if (isCompleteModelMessage(msg) && msg.payload.type === "tool_call_output") { + const id = (msg.payload as { tool_call_id: string }).tool_call_id; + if (firstCompleteAt[id] === undefined) firstCompleteAt[id] = Date.now() - start; + } + } + + // Both tools were approved. + expect(approvedAt["t1"]).toBeDefined(); + expect(approvedAt["t2"]).toBeDefined(); + // The second tool's approval happens before the first tool's execution completes (the slow + // command has not finished yet) -- i.e., execution does not block the next approval. + expect(approvedAt["t2"]!).toBeLessThan(firstCompleteAt["t1"] ?? Infinity); + // The fast b.txt finishes first, the slow a.txt finishes later (outputs in completion order). + expect(firstCompleteAt["t2"]!).toBeLessThan(firstCompleteAt["t1"]!); + + expect(await readFile(join(workspace, "a.txt"), "utf8")).toBe("one"); + expect(await readFile(join(workspace, "b.txt"), "utf8")).toBe("two"); + // Both tool outputs are fed back into the second turn, producing the final reply. + expect( + all.some( + (m) => + isCompleteModelMessage(m) && m.payload.type === "text" && m.payload.text === "both done", + ), + ).toBe(true); + }); + + it("collects all tool outputs (count matches tool calls) before the next round", async () => { + const llm = { + calls: 0, + async *streamGenerate(this: { calls: number }, params) { + this.calls += 1; + if (this.calls === 1) { + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "printf x" }), + toolCallId: "u1", + stopReason: "completed", + }); + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "printf y" }), + toolCallId: "u2", + stopReason: "completed", + }); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + } + // The second turn receives two tool_call_outputs. + const outputs = params.newMessages.filter( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + ); + expect(outputs).toHaveLength(2); + yield assistantText("ok"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + } as LLMInterface & { calls: number }; + + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment }); + + const all: OmniMessage[] = []; + for await (const msg of engine.run([userText("go")], { approve: allowAll })) { + all.push(msg); + } + const completeOutputs = all.filter( + (m) => isCompleteModelMessage(m) && m.payload.type === "tool_call_output", + ); + expect(completeOutputs).toHaveLength(2); + // The second turn did happen (otherwise the outputs assertion inside streamGenerate above + // would never run), and the "ok" it produces appears in the output stream. + expect(llm.calls).toBe(2); + expect( + all.some( + (m) => isCompleteModelMessage(m) && m.payload.type === "text" && m.payload.text === "ok", + ), + ).toBe(true); + }); +}); + +describe("ContextEngine tool execution resilience", () => { + let workspace: string; + beforeEach(async () => { + workspace = await mkdtemp(join(tmpdir(), "penguin-ws4-")); + }); + afterEach(async () => { + await rm(workspace, { recursive: true, force: true }); + }); + + it("feeds a failed tool output back and keeps tool_use/result paired (Environment converges errors, never throws)", async () => { + // The Environment converges the error into one complete tool_call_output (never throws); + // verify the engine feeds it back normally. + const failingEnv: EnvironmentInterface = { + async listTools() { + return []; + }, + toolPermission(): ToolPermission | undefined { + return "rw"; + }, + async *executeTool(request) { + const id = request.toolCall.payload.tool_call_id; + yield toolCallOutput({ + output: "[tool error] boom", + toolCallId: id, + stopReason: "failed", + }); + }, + }; + const llm: LLMInterface = { + calls: 0, + async *streamGenerate(this: { calls: number }, params) { + this.calls += 1; + const usage = () => + tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + if (this.calls === 1) { + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "x" }), + toolCallId: "e1", + stopReason: "completed", + }); + yield usage(); // The model output completes normally. + return { status: "completed" }; + } + // Second turn: must receive one tool_call_output (the failure reply), keeping the pairing. + const outputs = params.newMessages.filter( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + ); + expect(outputs).toHaveLength(1); + expect((outputs[0]!.payload as { output: string }).output).toContain("boom"); + yield assistantText("recovered"); + yield usage(); + return { status: "completed" }; + }, + } as LLMInterface & { calls: number }; + + const engine = new ContextEngine({ llm, environment: failingEnv }); + const all: OmniMessage[] = []; + for await (const msg of engine.run([userText("go")], { approve: allowAll })) { + all.push(msg); + } + // A failed failure output was produced, and the Task normally advanced to the second turn. + const failed = all.find( + (m) => + (m.payload as { type?: string }).type === "tool_call_output" && + (m.payload as { stop_reason?: string }).stop_reason === "failed", + ); + expect(failed).toBeDefined(); + expect( + all.some( + (m) => + isCompleteModelMessage(m) && m.payload.type === "text" && m.payload.text === "recovered", + ), + ).toBe(true); + }); + + it("converts a throwing executeTool (contract-violating custom environment) into a failed tool_call_output", async () => { + // EnvironmentInterface's contract says it never throws, but a custom implementation can be + // injected via the public API: a contract-violating exception must be converged by the + // engine's boundary safety net into a failed output (keeping tool_use/result paired), and + // must never become an unhandled rejection. + const throwingEnv: EnvironmentInterface = { + async listTools() { + return []; + }, + toolPermission(): ToolPermission | undefined { + return "rw"; + }, + // eslint-disable-next-line require-yield + async *executeTool(): AsyncGenerator { + throw new Error("custom env exploded"); + }, + }; + const llm: LLMInterface = { + calls: 0, + async *streamGenerate(this: { calls: number }, params) { + this.calls += 1; + const usage = () => + tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + if (this.calls === 1) { + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "x" }), + toolCallId: "t1", + stopReason: "completed", + }); + yield usage(); + return { status: "completed" }; + } + // Second turn: the contract-violating exception has been converged into a failed + // tool_call_output fed back, so the pairing is intact. + const outputs = params.newMessages.filter( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + ); + expect(outputs).toHaveLength(1); + expect((outputs[0]!.payload as { output: string }).output).toContain("custom env exploded"); + expect((outputs[0]!.payload as { stop_reason?: string }).stop_reason).toBe("failed"); + yield assistantText("survived"); + yield usage(); + return { status: "completed" }; + }, + } as LLMInterface & { calls: number }; + + const engine = new ContextEngine({ llm, environment: throwingEnv }); + const all: OmniMessage[] = []; + for await (const msg of engine.run([userText("go")], { approve: allowAll })) { + all.push(msg); + } + expect( + all.some( + (m) => + isCompleteModelMessage(m) && m.payload.type === "text" && m.payload.text === "survived", + ), + ).toBe(true); + }); + + it("treats a throwing approve callback as deny instead of letting the exception escape run", async () => { + const llm: LLMInterface = { + calls: 0, + async *streamGenerate(this: { calls: number }, params) { + this.calls += 1; + const usage = () => + tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + if (this.calls === 1) { + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "x" }), + toolCallId: "t1", + stopReason: "completed", + }); + yield usage(); + return { status: "completed" }; + } + // Second turn: the approval exception is converged to deny, feeding back one aborted + // output, keeping the pairing intact. + const outputs = params.newMessages.filter( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + ); + expect(outputs).toHaveLength(1); + expect((outputs[0]!.payload as { stop_reason?: string }).stop_reason).toBe("aborted"); + yield assistantText("done"); + yield usage(); + return { status: "completed" }; + }, + } as LLMInterface & { calls: number }; + + const engine = new ContextEngine({ + llm, + environment: new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }), + }); + const all: OmniMessage[] = []; + for await (const msg of engine.run([userText("go")], { + approve: async () => { + throw new Error("approval channel closed"); + }, + })) { + all.push(msg); + } + const denied = all.find( + (m) => + (m.payload as { type?: string }).type === "approval_decision" && + (m.payload as { decision?: string }).decision === "deny", + ); + expect(denied).toBeDefined(); + expect( + all.some( + (m) => isCompleteModelMessage(m) && m.payload.type === "text" && m.payload.text === "done", + ), + ).toBe(true); + }); +}); + +describe("ContextEngine abort during execution", () => { + let workspace: string; + beforeEach(async () => { + workspace = await mkdtemp(join(tmpdir(), "penguin-ws3-")); + }); + afterEach(async () => { + await rm(workspace, { recursive: true, force: true }); + }); + + it("aborting a long-running tool ends the turn, emits abort, and carries tool results over (model output completed)", async () => { + const received: OmniMessage[][] = []; + let call = 0; + const llm: LLMInterface = { + async *streamGenerate(params) { + received.push(params.newMessages); + call += 1; + if (call === 1) { + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "sleep 5" }), + toolCallId: "slow", + stopReason: "completed", + }); + // Model output completed (outcome=completed) -> AgentHub has committed this turn + // including the tool_call. + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + } + yield assistantText("resumed"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment }); + const controller = new AbortController(); + const startedAt = Date.now(); + setTimeout(() => controller.abort(), 200); + + const all: OmniMessage[] = []; + for await (const msg of engine.run([userText("go")], { + approve: allowAll, + signal: controller.signal, + })) { + all.push(msg); + } + const elapsed = Date.now() - startedAt; + + expect(elapsed).toBeLessThan(3000); // Did not wait the full 5s. + expect(all.map((m) => (m.payload as { type?: string }).type)).toContain("abort"); + + // Case A: model output has completed -> the interrupted tool's result is backfilled as a + // structured tool_call_output, pairing with the already-committed tool_call. + await collectRun(engine, [userText("continue")], allowAll); + expect(received).toHaveLength(2); + const out = received[1]!.find( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + ); + expect(out).toBeDefined(); + expect((out!.payload as { tool_call_id?: string }).tool_call_id).toBe("slow"); + // Case A must be a structured backfill and must **not** be flattened into + // (otherwise the already-committed tool_call would lose its pairing). + const carriedText = received[1]! + .map((m) => (m.payload as { text?: string }).text ?? "") + .join(""); + expect(carriedText).not.toContain(""); + }); + + it("case A backfills outputs for committed-but-undispatched tool_calls (preserves pairing)", async () => { + const received: OmniMessage[][] = []; + const controller = new AbortController(); + let call = 0; + const llm: LLMInterface = { + async *streamGenerate(params) { + received.push(params.newMessages); + call += 1; + if (call === 1) { + // Two real tool_calls + token_usage: AgentHub commits this turn including both + // a1 and a2 tool_calls. + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "true" }), + toolCallId: "a1", + stopReason: "completed", + }); + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "true" }), + toolCallId: "a2", + stopReason: "completed", + }); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + } + yield assistantText("resumed"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment }); + // Interrupted immediately while approving the first tool: after a1's approval, signal is + // already aborted -> neither a1 nor a2 is dispatched, but both have been committed by AgentHub. + let approvals = 0; + const approve: ApproveFn = async () => { + approvals += 1; + if (approvals === 1) controller.abort(); + return "allow"; + }; + + const all = await collectRun(engine, [userText("go")], approve, controller.signal); + expect(all.map((m) => (m.payload as { type?: string }).type)).toContain("abort"); + + // Next run: case A's structured carry-over must backfill paired outputs for both a1 and a2 + // (the undispatched a2 gets an interrupted-state placeholder). + await collectRun(engine, [userText("continue")], allowAll); + const ids = received[1]! + .filter((m) => (m.payload as { type?: string }).type === "tool_call_output") + .map((m) => (m.payload as { tool_call_id?: string }).tool_call_id); + expect(ids).toContain("a1"); + expect(ids).toContain("a2"); + }); +}); + +describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => { + let workspace: string; + beforeEach(async () => { + workspace = await mkdtemp(join(tmpdir(), "penguin-ws4-")); + }); + afterEach(async () => { + await rm(workspace, { recursive: true, force: true }); + }); + + it("auto-retries on LLM timeout: original input + carrying partial products", async () => { + let calls = 0; + const inputs: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + calls += 1; + inputs.push(params.newMessages); + if (calls === 1) { + // Timeout/network drop: produces partial text and ends without a token_usage, + // returning timeout. + yield partialText("start"); + yield partialText("delta", "thinking..."); + yield partialText("stop", "", "timeout"); + yield assistantText("thinking...", "timeout"); + return { status: "timeout" }; + } + // Retry succeeds: completes normally. + yield assistantText("done"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ + llm, + environment, + maxReconnects: 2, + reconnectBackoffMs: 0, + }); + + const all = await collectRun(engine, [userText("go")], allowAll); + + expect(calls).toBe(2); // Initial timeout -> auto-retries once within the same run and succeeds. + // Retry = original input kept as-is + one carrying the partial products + // already produced (not , to avoid the model mistaking it for a user interrupt). + expect(inputs[1]).toHaveLength(2); + expect(inputs[1]![0]).toEqual(inputs[0]![0]); + const retried = (inputs[1]![1]!.payload as { text?: string }).text ?? ""; + expect(retried).toContain(""); + expect(retried).toContain("thinking..."); + expect(retried).not.toContain(""); + // The final reply is produced, with no abort throughout. + expect( + all.some( + (m) => isCompleteModelMessage(m) && m.payload.type === "text" && m.payload.text === "done", + ), + ).toBe(true); + expect(all.map((m) => (m.payload as { type?: string }).type)).not.toContain("abort"); + }); + + it("skips a malformed (never-committed) tool_call: no dispatch, no paired output", async () => { + // A tool_call produced by an interrupted finish (stop_reason not completed) is never + // committed into history by AgentHub: the engine does not dispatch it for execution, does + // not add it to this turn's ledger, and does not backfill a paired output; the malformed + // turn is cleaned up by reconnect. + let calls = 0; + const inputs: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + calls += 1; + inputs.push(params.newMessages); + if (calls === 1) { + yield toolCall({ + name: "exec_command", + arguments: '{"cmd": "ec', + toolCallId: "tc-broken", + stopReason: "malformed", + }); + return { status: "malformed", message: "incomplete stream" }; + } + yield assistantText("done"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment, maxReconnects: 1, reconnectBackoffMs: 0 }); + + const all = await collectRun(engine, [userText("go")], allowAll); + + // The reconnect retry succeeds, with no abort throughout. + expect(calls).toBe(2); + expect(all.map((m) => (m.payload as { type?: string }).type)).not.toContain("abort"); + + // No output pointing to that tool_call is produced; the tool is also never approved/executed. + const paired = all.find( + (m) => + isCompleteModelMessage(m) && + m.payload.type === "tool_call_output" && + m.payload.tool_call_id === "tc-broken", + ); + expect(paired).toBeUndefined(); + expect(all.map((m) => (m.payload as { type?: string }).type)).not.toContain( + "approval_decision", + ); + + // The retry resends the original input as-is; the half-formed tool_call is discarded + // entirely and does not appear in the retry input in any form. + expect(inputs[1]).toEqual(inputs[0]); + const retryText = (inputs[1]![0]!.payload as { text?: string }).text ?? ""; + expect(retryText).toBe("go"); + }); + + it("auto-retries on LLM malformed: original input + carrying partial products", async () => { + let calls = 0; + const inputs: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + calls += 1; + inputs.push(params.newMessages); + if (calls === 1) { + yield partialText("start"); + yield partialText("delta", "partial json response"); + yield partialText("stop", "", "malformed"); + yield assistantText("partial json response", "malformed"); + return { + status: "malformed", + message: "Unexpected token < in JSON at position 0", + }; + } + yield assistantText("done"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ + llm, + environment, + maxReconnects: 1, + reconnectBackoffMs: 0, + }); + + const all = await collectRun(engine, [userText("go")], allowAll); + + expect(calls).toBe(2); + // The malformed attempt never entered AgentHub history: the original input is resent, + // plus carrying the partial products already produced. + expect(inputs[1]).toHaveLength(2); + expect(inputs[1]![0]).toEqual(inputs[0]![0]); + const retried = (inputs[1]![1]!.payload as { text?: string }).text ?? ""; + expect(retried).toContain(""); + expect(retried).toContain("partial json response"); + expect(retried).not.toContain(""); + expect( + all.some( + (m) => isCompleteModelMessage(m) && m.payload.type === "text" && m.payload.text === "done", + ), + ).toBe(true); + expect(all.map((m) => (m.payload as { type?: string }).type)).not.toContain("abort"); + }); + + it("emits abort and carries the original input over when reconnect retries are exhausted", async () => { + let calls = 0; + const inputs: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + calls += 1; + inputs.push(params.newMessages); + yield assistantText("partial...", "timeout"); + return { status: "timeout" }; // Always needs a reconnect. + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ + llm, + environment, + maxReconnects: 1, + reconnectBackoffMs: 0, + }); + + const all = await collectRun(engine, [userText("go")], allowAll); + expect(calls).toBe(2); // Initial attempt + maxReconnects(1) retries. + const abort = all.find((m) => (m.payload as { type?: string }).type === "abort"); + expect(abort).toBeDefined(); + expect((abort!.payload as { reason?: string }).reason).toContain("reconnect failed"); + + // carry-over = original input + (accumulating partial products from both + // failed attempts): the next run resends it merged with the new input, without producing + // . + await collectRun(engine, [userText("next")], allowAll); + const nextRunTexts = inputs[2]!.map((m) => (m.payload as { text?: string }).text ?? ""); + expect(nextRunTexts).toHaveLength(3); + expect(nextRunTexts[0]).toBe("go"); + expect(nextRunTexts[1]).toContain(""); + expect(nextRunTexts[1]).toContain("partial..."); + expect(nextRunTexts[2]).toBe("next"); + expect(nextRunTexts.join("\n")).not.toContain(""); + }); + + it("surfaces a non-retryable LLM failure (outcome=failed) as a graceful abort (run does not throw)", async () => { + let calls = 0; + const inputs: OmniMessage[][] = []; + const llm: LLMInterface = { + // The LLM must never throw an exception at the engine: a non-retryable error resolves + // by returning a failed outcome after closing the structure. + // eslint-disable-next-line require-yield + async *streamGenerate(params) { + calls += 1; + inputs.push(params.newMessages); + return { status: "failed", message: "invalid api key" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment, reconnectBackoffMs: 0 }); + + // Must not throw; should gracefully converge to an abort. + const all = await collectRun(engine, [userText("go")], allowAll); + expect(calls).toBe(1); // failed -> no retry. + const abort = all.find((m) => (m.payload as { type?: string }).type === "abort"); + expect(abort).toBeDefined(); + const reason = (abort!.payload as { reason?: string }).reason ?? ""; + expect(reason).toContain("llm request error"); + expect(reason).toContain("invalid api key"); + + // The failed turn's input is flattened and stashed; the next run resends it merged with + // the new input. + await collectRun(engine, [userText("next")], allowAll); + const text = inputs[1]!.map((m) => (m.payload as { text?: string }).text ?? "").join("\n"); + expect(text).toContain("go"); + expect(text).toContain("next"); + }); + + it("LLM timeout after a tool already executed: retry carries the call/result via (tool runs once)", async () => { + let calls = 0; + const inputs: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + calls += 1; + inputs.push(params.newMessages); + if (calls === 1) { + // Real tool_call -> the engine dispatches it for execution (appends to a file, a + // side effect), followed by a timeout/network drop (timeout). + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "printf x >> count.txt" }), + toolCallId: "t1", + stopReason: "completed", + }); + return { status: "timeout" }; + } + yield assistantText("second"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ + llm, + environment, + maxReconnects: 3, + reconnectBackoffMs: 0, + }); + + const all = await collectRun(engine, [userText("go")], allowAll); + + // The tool executes exactly once: retry input = original input + (containing + // a text transcript of the t1 call/result), so the model does not call it again; the + // transcript is plain text and is never dispatched again. + const content = await readFile(join(workspace, "count.txt"), "utf8").catch(() => ""); + expect(content).toBe("x"); + expect(calls).toBe(2); // Completes after one retry within the same run. + expect(inputs[1]![0]).toEqual(inputs[0]![0]); + const retried = (inputs[1]![1]!.payload as { text?: string }).text ?? ""; + expect(retried).toContain(""); + expect(retried).toContain(''); + expect(retried).toContain(' + isCompleteModelMessage(m) && m.payload.type === "text" && m.payload.text === "second", + ), + ).toBe(true); + expect(all.map((m) => (m.payload as { type?: string }).type)).not.toContain("abort"); + }); + + it("flatten carry-over (failed exit) includes the model's partial thinking and text (PRN-014)", async () => { + let calls = 0; + const inputs: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + calls += 1; + inputs.push(params.newMessages); + if (calls === 1) { + // Before the non-retryable error, partial thinking and text were already produced + // (the LLM finishes them as complete messages, stop_reason failed). + yield thinkingMessage("half-thought", "failed"); + yield assistantText("half-text", "failed"); + return { status: "failed", message: "boom" }; + } + yield assistantText("ok"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment, reconnectBackoffMs: 0 }); + + await collectRun(engine, [userText("go")], allowAll); + expect(calls).toBe(1); // failed -> no retry, exits immediately. + + // Next run: the flattened carry-over contains the original input plus partial thinking/text + // (both completed and incomplete messages are carried over). + await collectRun(engine, [userText("next")], allowAll); + const text = inputs[1]!.map((m) => (m.payload as { text?: string }).text ?? "").join("\n"); + expect(text).toContain(""); + expect(text).toContain("half-thought"); + expect(text).toContain("half-text"); + expect(text).toContain("go"); + expect(text).toContain("next"); + }); + + it("carry-over after exhausted retries: raw original input + with all attempts' products", async () => { + let calls = 0; + const inputs: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + calls += 1; + inputs.push(params.newMessages); + if (calls === 1) { + // Attempt 1: a real tool_call (execution has a side effect) followed by a timeout. + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "printf x >> chain.txt" }), + toolCallId: "t1", + stopReason: "completed", + }); + return { status: "timeout" }; + } + if (calls === 2) { + // Attempt 2 (retry, original input resent): produces partial thinking then times out + // again -> retries exhausted. + yield thinkingMessage("retry-thought", "timeout"); + return { status: "timeout" }; + } + yield assistantText("ok"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + // maxReconnects=1 -> exhausted after attempt 2; the original input is stashed as carry-over + // for the next run. + const engine = new ContextEngine({ + llm, + environment, + maxReconnects: 1, + reconnectBackoffMs: 0, + }); + + await collectRun(engine, [userText("go")], allowAll); + expect(calls).toBe(2); + // Retry = original input + (attempt 1's t1 call/result). + expect(inputs[1]![0]).toEqual(inputs[0]![0]); + expect((inputs[1]![1]!.payload as { text?: string }).text ?? "").toContain(""); + + // Next run: carry-over = original input + (accumulating attempt 1's t1 + // call/result and attempt 2's partial thinking), a single un-nested block; produces no + // . + await collectRun(engine, [userText("next")], allowAll); + const nextRunTexts = inputs[2]!.map((m) => (m.payload as { text?: string }).text ?? ""); + expect(nextRunTexts).toHaveLength(3); + expect(nextRunTexts[0]).toBe("go"); + const block = nextRunTexts[1]!; + expect(block).toContain(''); + expect(block).toContain('retry-thought"); + expect((block.match(//g) ?? []).length).toBe(1); + expect(nextRunTexts[2]).toBe("next"); + expect(nextRunTexts.join("\n")).not.toContain(""); + // t1 already executed once during the failed attempts (side effect occurred); the + // transcript is plain text and is not dispatched again by either the retry or the next run. + const content = await readFile(join(workspace, "chain.txt"), "utf8").catch(() => ""); + expect(content).toBe("x"); + }); + + it("user abort after a failed retry: un-nests into the flatten", async () => { + const controller = new AbortController(); + let calls = 0; + const inputs: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + calls += 1; + inputs.push(params.newMessages); + if (calls === 1) { + yield thinkingMessage("half-1", "timeout"); + return { status: "timeout" }; + } + if (calls === 2) { + // Interrupted by the user while the retry is in progress. + controller.abort(); + return { status: "aborted" }; + } + yield assistantText("ok"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ + llm, + environment, + maxReconnects: 2, + reconnectBackoffMs: 0, + }); + + await collectRun(engine, [userText("go")], allowAll, controller.signal); + expect(calls).toBe(2); + // The retry input carries . + expect((inputs[1]![1]!.payload as { text?: string }).text ?? "").toContain(""); + + // The next run after the interrupt: flattens into a single-level , with + // 's content un-nested and merged in. + await collectRun(engine, [userText("next")], allowAll); + const text = inputs[2]!.map((m) => (m.payload as { text?: string }).text ?? "").join("\n"); + expect(text).toContain(""); + expect(text).toContain("half-1"); + expect(text).not.toContain(""); + expect((text.match(//g) ?? []).length).toBe(1); + }); + + it("keeps raw inputs across repeated pre-request aborts (no flatten)", async () => { + const received: OmniMessage[][] = []; + const llm: LLMInterface = { + async *streamGenerate(params) { + received.push(params.newMessages); + yield assistantText("ok"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment }); + + // Run 1 & 2 are both interrupted before dispatch -> the input is stashed as-is as + // carry-over (run 1/2 never call the LLM). + const c1 = new AbortController(); + c1.abort(); + await collectRun(engine, [userText("go")], allowAll, c1.signal); + const c2 = new AbortController(); + c2.abort(); + await collectRun(engine, [userText("next")], allowAll, c2.signal); + + // Run 3 is normal: its first LLM input = the as-is preserved "go", "next" + "more", + // producing no (input that never made it to a Request is kept as-is + // per the trailing-input semantics). + await collectRun(engine, [userText("more")], allowAll); + const texts = received[0]!.map((m) => (m.payload as { text?: string }).text ?? ""); + expect(texts).toEqual(["go", "next", "more"]); + expect(texts.join("\n")).not.toContain(""); + }); +}); diff --git a/packages/core/test/environment.test.ts b/packages/core/test/environment.test.ts new file mode 100644 index 0000000..345fb2d --- /dev/null +++ b/packages/core/test/environment.test.ts @@ -0,0 +1,1013 @@ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { mkdtemp, readFile, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { Environment } from "../src/environment/index.js"; +import { + partialToolCallOutput, + toolCall, + toolCallOutput, + withOrigin, +} from "../src/omnimessage/index.js"; +import type { OmniMessage } from "../src/omnimessage/index.js"; +import { BUILTIN_TOOL_FACTORIES } from "../src/environment/tools/registry.js"; +import type { ToolConfig, ToolDefinitionConfig } from "../src/interfaces.js"; + +/** Tool config for exec_command (permission/maxOutputLength adjustable). */ +function execTool(overrides: Partial = {}): ToolDefinitionConfig { + return { + name: "exec_command", + description: "Run a shell command in the workspace.", + parameters: { + type: "object", + properties: { + cmd: { type: "string" }, + workdir: { type: "string" }, + }, + required: ["cmd"], + }, + permission: "rw", + maxOutputLength: 16000, + ...overrides, + }; +} + +function makeToolConfig(tool: ToolDefinitionConfig = execTool()): ToolConfig { + return { customTools: [tool], mcpServers: [] }; +} + +/** Collects all OmniMessages produced by an async generator. */ +async function collect(gen: AsyncGenerator): Promise { + const out: OmniMessage[] = []; + for await (const msg of gen) { + out.push(msg); + } + return out; +} + +function payloadTypes(messages: OmniMessage[]): string[] { + return messages.map((m) => (m.payload as { type?: string }).type ?? ""); +} + +let tmp: string; +let originalHome: string | undefined; + +beforeEach(async () => { + tmp = await mkdtemp(path.join(tmpdir(), "penguin-env-")); + // exec_command runs via a `bash -l` login shell (product behavior): a login shell loads the + // developer's ~/.bash_profile and similar files, whose latency (e.g. nvm taking hundreds of + // ms) and stderr output (e.g. nvm warnings) can leak into tool output, letting the local + // profile hijack timeout/truncation test cases. Pointing HOME at an empty temp directory makes + // the login shell read only the system-level profile (quiet, millisecond-scale), decoupling + // tests from the developer's environment. + originalHome = process.env.HOME; + process.env.HOME = tmp; +}); + +afterEach(async () => { + if (originalHome === undefined) delete process.env.HOME; + else process.env.HOME = originalHome; + await rm(tmp, { recursive: true, force: true }); +}); + +describe("Environment.listTools", () => { + it("returns exactly one exec_command tool with only definition fields", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(), + }); + const tools = await env.listTools(); + expect(tools).toHaveLength(1); + // Deep-equal to exactly the definition fields -- also proves permission / maxOutputLength + // do not leak into the LLM tool definition. + expect(tools[0]).toEqual({ + name: "exec_command", + description: "Run a shell command in the workspace.", + parameters: { + type: "object", + properties: { + cmd: { type: "string" }, + workdir: { type: "string" }, + }, + required: ["cmd"], + }, + }); + }); + + it("does not expose configured tools that are not supported by the registry", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + customTools: [ + execTool(), + { name: "not_a_registered_tool", description: "unsupported", permission: "r" }, + ], + mcpServers: [], + }, + }); + // An unrecognized tool name is neither executable nor exposed to the LLM. + const tools = await env.listTools(); + expect(tools.map((t) => t.name)).toEqual(["exec_command"]); + }); +}); + +describe("Environment.executeTool — basic file write", () => { + it("streams start/delta?/stop + final tool_call_output and writes the file", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(), + }); + const call = toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "printf 'Hello, Penguin' > note.txt" }), + toolCallId: "call_write", + }); + + const messages = await collect(env.executeTool({ toolCall: call })); + + const types = payloadTypes(messages); + // The first is partial(start), includes one partial(stop), and the last is the complete + // tool_call_output. + expect(types[0]).toBe("partial_tool_call_output"); + expect((messages[0]!.payload as { event_type?: string }).event_type).toBe("start"); + expect(types).toContain("partial_tool_call_output"); + const last = messages[messages.length - 1]!; + expect((last.payload as { type?: string }).type).toBe("tool_call_output"); + + // There is a stop partial. + const hasStop = messages.some( + (m) => + (m.payload as { type?: string }).type === "partial_tool_call_output" && + (m.payload as { event_type?: string }).event_type === "stop", + ); + expect(hasStop).toBe(true); + + // tool_call_id is echoed back as-is; a successful command has stop_reason completed. + const outPayload = last.payload as { + tool_call_id: string; + stop_reason?: string; + }; + expect(outPayload.tool_call_id).toBe("call_write"); + expect(outPayload.stop_reason).toBe("completed"); + + const written = await readFile(path.join(tmp, "note.txt"), "utf8"); + expect(written).toBe("Hello, Penguin"); + }); +}); + +describe("Environment.executeTool — vault env injection", () => { + it("injects vault entries into the command env; hardened entries are not overridable", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(), + // PAGER is a hardened entry (HARDENED_ENV); a same-named vault entry must not override it. + vault: { PENGUIN_VAULT_TEST_KEY: "vault-secret-value", PAGER: "less" }, + }); + const call = toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: 'echo "k=$PENGUIN_VAULT_TEST_KEY pager=$PAGER"' }), + toolCallId: "call_vault", + }); + + const messages = await collect(env.executeTool({ toolCall: call })); + const last = messages[messages.length - 1]!.payload as { output?: string }; + expect(last.output).toContain("k=vault-secret-value"); + // Injection order is vault -> HARDENED_ENV: the hardened entry wins (settings that prevent + // an interactive hang must not be overridable). + expect(last.output).toContain("pager=cat"); + env.dispose(); + }); + + it("leaves the command env untouched when no vault is configured", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(), + }); + const call = toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: 'echo "k=[$PENGUIN_VAULT_TEST_KEY]"' }), + toolCallId: "call_no_vault", + }); + + const messages = await collect(env.executeTool({ toolCall: call })); + const last = messages[messages.length - 1]!.payload as { output?: string }; + expect(last.output).toContain("k=[]"); + env.dispose(); + }); +}); + +describe("Environment.executeTool — edit file", () => { + it("appends to an existing file", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(), + }); + + await collect( + env.executeTool({ + toolCall: toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "printf 'Hello' > note.txt" }), + toolCallId: "c1", + }), + }), + ); + await collect( + env.executeTool({ + toolCall: toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "printf '!' >> note.txt" }), + toolCallId: "c2", + }), + }), + ); + + const written = await readFile(path.join(tmp, "note.txt"), "utf8"); + expect(written).toBe("Hello!"); + }); +}); + +describe("Environment.executeTool — maxOutputLength truncation", () => { + it("truncates front-to-back at the limit with a trailing marker; stream == complete", async () => { + const maxOutputLength = 50; + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(execTool({ maxOutputLength })), + }); + + const messages = await collect( + env.executeTool({ + toolCall: toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "seq 1 100000" }), + toolCallId: "call_big", + }), + }), + ); + + const last = messages[messages.length - 1]!; + expect((last.payload as { type?: string }).type).toBe("tool_call_output"); + const output = (last.payload as { output: string }).output; + // Truncates front-to-back: the head is kept, and the truncation marker is appended at the + // tail (the marker does not count toward the limit). + expect(output.startsWith("1\n2\n3\n")).toBe(true); + const marker = `[output truncated: exceeded ${maxOutputLength} chars]`; + expect(output).toContain(marker); + expect(output.length).toBeLessThanOrEqual(maxOutputLength + marker.length + 1); + // Even when truncated, concatenating the streamed deltas == the complete content (the + // excess part is never forwarded). + const streamed = messages + .filter( + (m) => + (m.payload as { type?: string }).type === "partial_tool_call_output" && + (m.payload as { event_type?: string }).event_type === "delta", + ) + .map((m) => (m.payload as { output?: string }).output ?? "") + .join(""); + expect(streamed).toBe(output); + }); + + it("maxOutputLength <= 0 disables truncation", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(execTool({ maxOutputLength: 0 })), + }); + const messages = await collect( + env.executeTool({ + toolCall: toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "seq 1 100" }), + toolCallId: "call_nolimit", + }), + }), + ); + const output = (messages[messages.length - 1]!.payload as { output: string }).output; + expect(output).toContain("100"); + expect(output).not.toContain("[output truncated"); + }); +}); + +describe("Environment.executeTool — relaxed tool contract", () => { + it("frames a tool that yields bare deltas (no start/stop/complete) and reports via return value", async () => { + const NAME = "__bare_delta_tool__"; + BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({ + name: NAME, + definition, + async *execute(_args, ctx) { + // New contract: yields only deltas, with no start/stop, no complete message; the + // finish reason is reported via the return value. + yield partialToolCallOutput({ + eventType: "delta", + output: "partial ", + toolCallId: ctx.toolCallId, + }); + yield partialToolCallOutput({ + eventType: "delta", + output: "result", + toolCallId: ctx.toolCallId, + }); + return { stopReason: "failed" as const }; + }, + }); + try { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + customTools: [{ name: NAME, description: "bare", permission: "rw" }], + mcpServers: [], + }, + }); + const out = await collect( + env.executeTool({ + toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "b1" }), + }), + ); + // Environment uniformly frames it: start -> delta* -> stop -> complete message. + expect((out[0]!.payload as { event_type?: string }).event_type).toBe("start"); + const complete = out[out.length - 1]!.payload as { + type?: string; + output?: string; + stop_reason?: string; + }; + expect(complete.type).toBe("tool_call_output"); + expect(complete.output).toBe("partial result"); + expect(complete.stop_reason).toBe("failed"); // Finish reason reported via the return value + expect((out[out.length - 2]!.payload as { event_type?: string }).event_type).toBe("stop"); + } finally { + delete BUILTIN_TOOL_FACTORIES[NAME]; + } + }); + + it("carries ToolResult.images once via a streamed delta before stop, then again on the complete message", async () => { + const NAME = "__image_tool__"; + const dataUrl = "data:image/png;base64,AAAA"; + BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({ + name: NAME, + definition, + async *execute(_args, ctx) { + // Images are reported via the return value; text deltas stream as usual. + yield partialToolCallOutput({ + eventType: "delta", + output: "image/png, 4 B", + toolCallId: ctx.toolCallId, + }); + return { images: [dataUrl] }; + }, + }); + try { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + customTools: [{ name: NAME, description: "img", permission: "r" }], + mcpServers: [], + }, + }); + const out = await collect( + env.executeTool({ + toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "i1" }), + }), + ); + const complete = out[out.length - 1]!.payload as { + type?: string; + output?: string; + images?: string[]; + stop_reason?: string; + }; + expect(complete.type).toBe("tool_call_output"); + expect(complete.stop_reason).toBe("completed"); + expect(complete.output).toBe("image/png, 4 B"); + expect(complete.images).toEqual([dataUrl]); + // Streamed concatenation == complete message: images are not delta-streamed; they are + // carried once, whole, by a single delta right before stop. + const partials = out + .map((m) => m.payload as { type?: string; event_type?: string; images?: string[] }) + .filter((p) => p.type === "partial_tool_call_output"); + const withImages = partials.filter((p) => p.images !== undefined); + expect(withImages).toHaveLength(1); + expect(withImages[0]!.event_type).toBe("delta"); + expect(withImages[0]!.images).toEqual([dataUrl]); + // The image delta comes immediately before stop (after the text delta). + expect(partials[partials.length - 1]!.event_type).toBe("stop"); + expect(partials[partials.length - 2]!.images).toEqual([dataUrl]); + } finally { + delete BUILTIN_TOOL_FACTORIES[NAME]; + } + }); + + it("drops ToolResult.images when the tool did not complete normally", async () => { + const NAME = "__failed_image_tool__"; + BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({ + name: NAME, + definition, + async *execute(_args, ctx) { + yield partialToolCallOutput({ + eventType: "delta", + output: "broken", + toolCallId: ctx.toolCallId, + }); + return { stopReason: "failed" as const, images: ["data:image/png;base64,AAAA"] }; + }, + }); + try { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + customTools: [{ name: NAME, description: "img", permission: "r" }], + mcpServers: [], + }, + }); + const out = await collect( + env.executeTool({ + toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "i2" }), + }), + ); + const complete = out[out.length - 1]!.payload as { stop_reason?: string; images?: string[] }; + // Only a normal completion carries images: a failed finish drops them (neither the + // stream nor the complete message carries them), keeping the finish handling simple. + expect(complete.stop_reason).toBe("failed"); + expect(complete.images).toBeUndefined(); + for (const m of out) { + const p = m.payload as { type?: string }; + if (p.type === "partial_tool_call_output") expect("images" in p).toBe(false); + } + } finally { + delete BUILTIN_TOOL_FACTORIES[NAME]; + } + }); + + it("passes origin-tagged nested messages through verbatim, excluded from the tool output", async () => { + const NAME = "__forwarding_tool__"; + const hop = "sess_child"; + BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({ + name: NAME, + definition, + async *execute(_args, ctx) { + // Nested forwarding: origin-tagged messages pass through verbatim (a child session's + // complete tool_call_output is not folded into the finish either). + yield withOrigin(toolCallOutput({ output: "child result", toolCallId: "child_call" }), hop); + yield partialToolCallOutput({ + eventType: "delta", + output: "own output", + toolCallId: ctx.toolCallId, + }); + }, + }); + try { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + customTools: [{ name: NAME, description: "fwd", permission: "rw" }], + mcpServers: [], + }, + }); + const out = await collect( + env.executeTool({ + toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "f1" }), + }), + ); + // The forwarded nested message keeps its origin and original payload. + const forwarded = out.find((m) => m.origin?.length); + expect(forwarded).toBeDefined(); + expect((forwarded!.payload as { output?: string }).output).toBe("child result"); + // This tool's own complete output contains only its own deltas, not mixed with the + // child session's content. + const completes = out.filter( + (m) => !m.origin?.length && (m.payload as { type?: string }).type === "tool_call_output", + ); + expect(completes).toHaveLength(1); + expect((completes[0]!.payload as { output?: string }).output).toBe("own output"); + expect((completes[0]!.payload as { stop_reason?: string }).stop_reason).toBe("completed"); + } finally { + delete BUILTIN_TOOL_FACTORIES[NAME]; + } + }); +}); + +describe("Environment.executeTool — timeoutMs (PRN-013)", () => { + it("fails a tool exceeding timeoutMs, keeps prior output, and streams the timeout reason", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(execTool({ timeoutMs: 200 })), + }); + const startedAt = Date.now(); + + const messages = await collect( + env.executeTool({ + toolCall: toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "echo begin; sleep 5" }), + toolCallId: "call_timeout", + }), + }), + ); + + const elapsedMs = Date.now() - startedAt; + expect(elapsedMs).toBeLessThan(3000); // Did not wait the full 5s -> timeout aborts execution + + // A timeout is a failure: stop_reason failed, the timeout reason is written into + // tool_call_output, and the already-produced output is kept. + const last = messages[messages.length - 1]!.payload as { + type: string; + output: string; + stop_reason?: string; + }; + expect(last.type).toBe("tool_call_output"); + expect(last.stop_reason).toBe("failed"); + expect(last.output).toContain("begin"); + expect(last.output).toContain("[tool timeout: exceeded 200ms]"); + // The timeout marker is also produced via streaming: concatenating the streamed deltas == + // the complete content. + const streamed = messages + .filter( + (m) => + (m.payload as { type?: string }).type === "partial_tool_call_output" && + (m.payload as { event_type?: string }).event_type === "delta", + ) + .map((m) => (m.payload as { output?: string }).output ?? "") + .join(""); + expect(streamed).toBe(last.output); + }); + + it("user abort takes precedence over a pending timeout (aborted, not failed)", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(execTool({ timeoutMs: 60000 })), + }); + const controller = new AbortController(); + const messagesPromise = collect( + env.executeTool({ + toolCall: toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "sleep 5" }), + toolCallId: "call_user_abort", + }), + signal: controller.signal, + }), + ); + const abortTimer = setTimeout(() => controller.abort(), 100); + const messages = await messagesPromise; + clearTimeout(abortTimer); + + const last = messages[messages.length - 1]!.payload as { + output: string; + stop_reason?: string; + }; + expect(last.stop_reason).toBe("aborted"); + expect(last.output).toContain("[interrupted: tool aborted by user]"); + }); +}); + +describe("Environment.executeTool — robustness", () => { + it("returns an explanatory output for an unknown tool name without throwing", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(), + }); + + const messages = await collect( + env.executeTool({ + toolCall: toolCall({ + name: "not_a_real_tool", + arguments: "{}", + toolCallId: "call_unknown", + }), + }), + ); + + // Errors are also produced via streaming (renderable by the frontend): start -> + // delta(explanation) -> stop -> complete message. + expect(payloadTypes(messages)).toEqual([ + "partial_tool_call_output", + "partial_tool_call_output", + "partial_tool_call_output", + "tool_call_output", + ]); + const delta = messages[1]!.payload as { event_type?: string; output?: string }; + expect(delta.event_type).toBe("delta"); + expect(delta.output).toContain("Unknown tool: not_a_real_tool"); // Streamed content includes the explanation + const payload = messages[3]!.payload as { + type: string; + output: string; + tool_call_id: string; + stop_reason?: string; + }; + expect(payload.type).toBe("tool_call_output"); + expect(payload.output).toContain("Unknown tool: not_a_real_tool"); // Complete content matches + expect(payload.tool_call_id).toBe("call_unknown"); + expect(payload.stop_reason).toBe("failed"); + }); + + /** Runs one tool call and returns the payload of the last complete tool_call_output. */ + async function runTool(args: string, toolCallId: string) { + const env = new Environment({ workspaceDir: tmp, toolConfig: makeToolConfig() }); + const messages = await collect( + env.executeTool({ + toolCall: toolCall({ name: "exec_command", arguments: args, toolCallId }), + }), + ); + return messages[messages.length - 1]!.payload as { + type: string; + output: string; + stop_reason?: string; + }; + } + + it("converges unparsable arguments to an explanatory failed output without throwing", async () => { + // On the normal path, bad JSON already finishes as malformed at the LLM layer and is + // retried via reconnect, so it never reaches the Environment; this is the public interface's + // defensive fallback, uniformly converging to a "not valid JSON" failed output. + const bad = ["{not valid json", "{'a':1}", '{"a" "b"}', '{"cmd": "echo hi']; + for (const [i, args] of bad.entries()) { + const payload = await runTool(args, `call_badjson_${i}`); + expect(payload.type).toBe("tool_call_output"); + expect(payload.output, args).toContain("not valid JSON"); + expect(payload.stop_reason).toBe("failed"); + } + }); + + it("tells the model the arguments were empty", async () => { + const payload = await runTool("", "call_emptyargs"); + expect(payload.output).toContain("arguments field is empty"); + expect(payload.stop_reason).toBe("failed"); + }); + + it("never returns an empty tool output", async () => { + // A silently successful command (no stdout/stderr): an empty tool_result would leave the + // model unable to tell "no output" from "failure". + const payload = await runTool(JSON.stringify({ cmd: "true" }), "call_silent"); + expect(payload.type).toBe("tool_call_output"); + expect(payload.stop_reason).toBe("completed"); + expect(payload.output).toBe("[no output]"); + }); + + it("returns an explanatory output when cmd is missing", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(), + }); + + const messages = await collect( + env.executeTool({ + toolCall: toolCall({ + name: "exec_command", + arguments: JSON.stringify({ workdir: "." }), + toolCallId: "call_nocmd", + }), + }), + ); + + const last = messages[messages.length - 1]!; + const payload = last.payload as { type: string; output: string }; + expect(payload.type).toBe("tool_call_output"); + expect(payload.output).toContain("Missing required argument"); + }); + + it("reports a non-zero exit code in the final output", async () => { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: makeToolConfig(), + }); + + const messages = await collect( + env.executeTool({ + toolCall: toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "exit 3" }), + toolCallId: "call_fail", + }), + }), + ); + + const last = messages[messages.length - 1]!; + const payload = last.payload as { + output: string; + stop_reason?: string; + }; + expect(payload.output).toContain("[exit code: 3]"); + expect(payload.stop_reason).toBe("failed"); + // The exit-code marker is also produced via streaming (renderable by the frontend); the + // streamed deltas concatenated == the complete content (short output here, no truncation). + const streamed = messages + .filter( + (m) => + (m.payload as { type?: string }).type === "partial_tool_call_output" && + (m.payload as { event_type?: string }).event_type === "delta", + ) + .map((m) => (m.payload as { output?: string }).output ?? "") + .join(""); + expect(streamed).toContain("[exit code: 3]"); + expect(streamed).toBe(payload.output); + }); + + it("aborts background children without waiting for inherited pipes", async () => { + // #23: on interrupt, kill the whole process group -- even though background children + // inherit the stdout/stderr pipes, they should end immediately on abort rather than + // waiting for a natural exit (otherwise executeTool would be stuck on unclosed pipes). + const env = new Environment({ workspaceDir: tmp, toolConfig: makeToolConfig() }); + const controller = new AbortController(); + const startedAt = Date.now(); + + const messagesPromise = collect( + env.executeTool({ + toolCall: toolCall({ + name: "exec_command", + arguments: JSON.stringify({ + cmd: 'node -e "setTimeout(()=>{},5000)" & wait', + }), + toolCallId: "call_abort_bg", + }), + signal: controller.signal, + }), + ); + const abortTimer = setTimeout(() => controller.abort(), 200); + const messages = await messagesPromise; + clearTimeout(abortTimer); + + const elapsedMs = Date.now() - startedAt; + const last = messages[messages.length - 1]!; + const payload = last.payload as { output: string; stop_reason?: string }; + expect(elapsedMs).toBeLessThan(2000); // Did not wait the full 5s -> the process group was interrupted as a whole + expect(payload.output).toContain("[interrupted: tool aborted by user"); + expect(payload.stop_reason).toBe("aborted"); + }); +}); + +describe("Environment.toolPermission", () => { + it("returns the configured permission for a known tool", () => { + const env = new Environment({ + workspaceDir: "/tmp", + toolConfig: makeToolConfig(execTool({ permission: "rw" })), + }); + expect(env.toolPermission("exec_command")).toBe("rw"); + }); + + it("returns undefined for an unknown tool", () => { + const env = new Environment({ workspaceDir: "/tmp", toolConfig: makeToolConfig() }); + expect(env.toolPermission("nope")).toBeUndefined(); + }); +}); + +describe("Environment structure invariant on tool throw (PRN-012)", () => { + it("closes an open partial segment before the failed output when a tool throws mid-stream", async () => { + // Temporarily register a tool that yields partial(start)+delta then throws, to verify + // Environment backfills a partial(stop). + const NAME = "__throwing_test_tool__"; + BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({ + name: NAME, + definition, + async *execute(_args, ctx) { + yield partialToolCallOutput({ eventType: "start", toolCallId: ctx.toolCallId }); + yield partialToolCallOutput({ + eventType: "delta", + output: "working", + toolCallId: ctx.toolCallId, + }); + throw new Error("kaboom"); + }, + }); + try { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + customTools: [{ name: NAME, description: "throws", permission: "rw" }], + mcpServers: [], + }, + }); + const out = await collect( + env.executeTool({ + toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "z1" }), + }), + ); + // start -> delta(working) -> delta(error marker) -> stop -> complete failed output: the + // error marker is also produced via streaming, and the concatenated streamed fragments + // match the complete message. + expect( + out.map( + (m) => + `${(m.payload as { type?: string }).type}:${(m.payload as { event_type?: string }).event_type ?? ""}`, + ), + ).toEqual([ + "partial_tool_call_output:start", + "partial_tool_call_output:delta", + "partial_tool_call_output:delta", + "partial_tool_call_output:stop", + "tool_call_output:", + ]); + const noteDelta = out[2]!.payload as { output?: string }; + expect(noteDelta.output).toContain("kaboom"); // The error marker is produced via a streamed delta + const stop = out[3]!.payload as { stop_reason?: string }; + expect(stop.stop_reason).toBe("failed"); + const complete = out[4]!.payload as { stop_reason?: string; output?: string }; + expect(complete.stop_reason).toBe("failed"); + expect(complete.output).toContain("kaboom"); + expect(complete.output).toContain("working"); // Keeps the partial content already streamed. + // Concatenating the streamed deltas == the complete content (relevant for frontend rendering). + const streamedText = out + .filter( + (m) => + (m.payload as { type?: string }).type === "partial_tool_call_output" && + (m.payload as { event_type?: string }).event_type === "delta", + ) + .map((m) => (m.payload as { output?: string }).output ?? "") + .join(""); + expect(streamedText).toBe(complete.output); + } finally { + delete BUILTIN_TOOL_FACTORIES[NAME]; + } + }); +}); + +describe("Environment abort handling (打断 → aborted, PRN-012)", () => { + it("labels a thrown error as aborted (not failed) when the signal is aborted", async () => { + const NAME = "__abort_throw_tool__"; + BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({ + name: NAME, + definition, + async *execute(_args, ctx) { + yield partialToolCallOutput({ eventType: "start", toolCallId: ctx.toolCallId }); + yield partialToolCallOutput({ + eventType: "delta", + output: "partial", + toolCallId: ctx.toolCallId, + }); + // Simulates a throw caused by an interrupt (e.g. an underlying operation throwing AbortError). + throw new Error("aborted mid-run"); + }, + }); + try { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + customTools: [{ name: NAME, description: "throws", permission: "rw" }], + mcpServers: [], + }, + }); + const controller = new AbortController(); + controller.abort(); + const out = await collect( + env.executeTool({ + toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "z1" }), + signal: controller.signal, + }), + ); + // The structure is closed, and the interrupt maps to aborted (crucially: not failed). + const stop = out.find( + (m) => + (m.payload as { type?: string }).type === "partial_tool_call_output" && + (m.payload as { event_type?: string }).event_type === "stop", + )!.payload as { stop_reason?: string }; + expect(stop.stop_reason).toBe("aborted"); + const complete = out.find( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + )!.payload as { stop_reason?: string; output?: string }; + expect(complete.stop_reason).toBe("aborted"); + expect(complete.output).toContain("interrupted"); + expect(complete.output).toContain("partial"); // Keeps the partial content already streamed. + } finally { + delete BUILTIN_TOOL_FACTORIES[NAME]; + } + }); + + it("relabels a completed output as aborted when the signal is aborted, even if the tool did not self-report it", async () => { + const NAME = "__abort_noselfreport_tool__"; + BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({ + name: NAME, + definition, + async *execute(_args, ctx) { + // The tool does not self-report aborted, and only yields one ordinary complete output. + yield toolCallOutput({ output: "done anyway", toolCallId: ctx.toolCallId }); + }, + }); + try { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + customTools: [{ name: NAME, description: "ok", permission: "rw" }], + mcpServers: [], + }, + }); + const controller = new AbortController(); + controller.abort(); + const out = await collect( + env.executeTool({ + toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "z2" }), + signal: controller.signal, + }), + ); + const complete = out.find( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + )!.payload as { stop_reason?: string; output?: string }; + // Environment finalizes aborted based on the signal it holds, keeping the tool's + // already-produced content and appending the interrupt notice. + expect(complete.stop_reason).toBe("aborted"); + expect(complete.output).toContain("done anyway"); + expect(complete.output).toContain("interrupted"); + // Even when the tool yields only a complete message (no streaming), the whole content is + // backfilled as a stream: concatenating the streamed deltas == the complete content. + const streamed = out + .filter( + (m) => + (m.payload as { type?: string }).type === "partial_tool_call_output" && + (m.payload as { event_type?: string }).event_type === "delta", + ) + .map((m) => (m.payload as { output?: string }).output ?? "") + .join(""); + expect(streamed).toBe(complete.output); + } finally { + delete BUILTIN_TOOL_FACTORIES[NAME]; + } + }); + + it("streams full content when a tool emits start + a content-bearing complete but no delta, then is aborted (stream == complete, no separator drift)", async () => { + const NAME = "__abort_bufferonly_tool__"; + BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({ + name: NAME, + definition, + async *execute(_args, ctx) { + // An internally-buffering tool: yields start, produces no delta, and directly gives + // one complete message with content. + yield partialToolCallOutput({ eventType: "start", toolCallId: ctx.toolCallId }); + yield toolCallOutput({ output: "buffered result", toolCallId: ctx.toolCallId }); + }, + }); + try { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + customTools: [{ name: NAME, description: "x", permission: "rw" }], + mcpServers: [], + }, + }); + const controller = new AbortController(); + controller.abort(); + const out = await collect( + env.executeTool({ + toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "z3" }), + signal: controller.signal, + }), + ); + const complete = out.find( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + )!.payload as { stop_reason?: string; output?: string }; + expect(complete.stop_reason).toBe("aborted"); + expect(complete.output).toContain("buffered result"); + expect(complete.output).toContain("interrupted"); + // Key point: tool content that was never streamed is backfilled as a whole; concatenating + // the streamed deltas == the complete content (no separator misalignment). + const streamed = out + .filter( + (m) => + (m.payload as { type?: string }).type === "partial_tool_call_output" && + (m.payload as { event_type?: string }).event_type === "delta", + ) + .map((m) => (m.payload as { output?: string }).output ?? "") + .join(""); + expect(streamed).toBe(complete.output); + } finally { + delete BUILTIN_TOOL_FACTORIES[NAME]; + } + }); + + it("emits exactly one complete output (== streamed) with a trailing stop when a tool yields only partials and returns", async () => { + const NAME = "__no_complete_tool__"; + BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({ + name: NAME, + definition, + async *execute(_args, ctx) { + yield partialToolCallOutput({ eventType: "start", toolCallId: ctx.toolCallId }); + yield partialToolCallOutput({ + eventType: "delta", + output: "partial only", + toolCallId: ctx.toolCallId, + }); + // Does not yield a complete tool_call_output, and just returns (fallback path). + }, + }); + try { + const env = new Environment({ + workspaceDir: tmp, + toolConfig: { + customTools: [{ name: NAME, description: "x", permission: "rw" }], + mcpServers: [], + }, + }); + const out = await collect( + env.executeTool({ + toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "z4" }), + }), + ); + // The fallback still guarantees exactly one complete tool_call_output (keeping + // tool_use/result paired), with content == what was already streamed. + const completes = out.filter( + (m) => (m.payload as { type?: string }).type === "tool_call_output", + ); + expect(completes).toHaveLength(1); + expect((completes[0]!.payload as { output?: string }).output).toBe("partial only"); + // The last is the complete message, immediately preceded by a stop. + expect((out[out.length - 1]!.payload as { type?: string }).type).toBe("tool_call_output"); + expect((out[out.length - 2]!.payload as { event_type?: string }).event_type).toBe("stop"); + } finally { + delete BUILTIN_TOOL_FACTORIES[NAME]; + } + }); +}); diff --git a/packages/core/test/example-benchmark.test.ts b/packages/core/test/example-benchmark.test.ts new file mode 100644 index 0000000..168154b --- /dev/null +++ b/packages/core/test/example-benchmark.test.ts @@ -0,0 +1,163 @@ +/** + * Example Benchmark provisioning tests: + * default_agent initialization pre-seeds benchmarks/example-benchmark/ (a parseable config, + * runs=2, a scoreboard with three self-consistent evaluations); an ordinary Agent gets none; + * provisioning is skipped idempotently when benchmarks/ already exists. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { parse as parseToml } from "smol-toml"; +import { parse as parseYaml } from "yaml"; +import { + DEFAULT_AGENT_ID, + DEFAULT_PROJECT_ID, + EXAMPLE_BENCHMARK_ID, + benchmarksDir, + buildExampleScoreboard, + loadOrInitAgentState, + provisionProjectAgents, +} from "../src/state/index.js"; + +let tmpRoot: string; +let prevHome: string | undefined; + +beforeEach(async () => { + prevHome = process.env.PENGUIN_HOME; + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-bench-")); + process.env.PENGUIN_HOME = tmpRoot; +}); + +afterEach(async () => { + if (prevHome === undefined) delete process.env.PENGUIN_HOME; + else process.env.PENGUIN_HOME = prevHome; + await fs.rm(tmpRoot, { recursive: true, force: true }); +}); + +async function exists(p: string): Promise { + try { + await fs.access(p); + return true; + } catch { + return false; + } +} + +interface RunScore { + score: number; + cost: number; + duration_ms: number; + session_id: string; +} +interface CaseScore extends Omit { + case: string; + runs: RunScore[]; +} +interface Evaluation extends Omit { + time: string; + version: number; + provider: string; + model_id: string; + summary_title: string; + summary: string; + cases: CaseScore[]; +} + +describe("example benchmark provisioning", () => { + it("default_agent init creates a parseable example benchmark (config + scoreboard + cases)", async () => { + await loadOrInitAgentState(); + const dir = path.join( + benchmarksDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + EXAMPLE_BENCHMARK_ID, + ); + + // benchmark_config.toml: title/description/runs=2; contains no model reference (the model + // is recorded on each evaluation instead). + const config = parseToml(await fs.readFile(path.join(dir, "benchmark_config.toml"), "utf8")); + expect(config.title).toBe("Example Benchmark"); + expect(String(config.description)).toContain("built-in example"); + expect(String(config.description)).toContain("Replace it with your own"); + expect(Number(config.runs)).toBe(2); + expect(config).not.toHaveProperty("provider"); + expect(config).not.toHaveProperty("model_id"); + + // Two cases: statement/ and rubric/ each use README.md as the index. + for (const caseId of ["CASE-001-file-summary", "CASE-002-data-cleanup"]) { + const statement = await fs.readFile(path.join(dir, caseId, "statement", "README.md"), "utf8"); + const rubric = await fs.readFile(path.join(dir, caseId, "rubric", "README.md"), "utf8"); + expect(statement.length).toBeGreaterThan(50); + expect(rubric).toContain("pts"); + } + + // scoreboard.yaml: 3 evaluations, with version/time increasing and scores rising, and + // 2 runs per case. + const scoreboard = parseYaml(await fs.readFile(path.join(dir, "scoreboard.yaml"), "utf8")) as { + evaluations: Evaluation[]; + }; + expect(scoreboard.evaluations).toHaveLength(3); + expect(scoreboard.evaluations.map((e) => e.version)).toEqual([1, 2, 3]); + const times = scoreboard.evaluations.map((e) => new Date(e.time).getTime()); + expect(times[0]!).toBeLessThan(times[1]!); + expect(times[1]!).toBeLessThan(times[2]!); + const scores = scoreboard.evaluations.map((e) => e.score); + expect(scores[0]!).toBeLessThan(scores[1]!); + expect(scores[1]!).toBeLessThan(scores[2]!); + // Each evaluation carries the actual model used for that run (as a pair) and a summary + // title; the example consistently uses deepseek-v4-pro. + expect(scoreboard.evaluations.map((e) => e.model_id)).toEqual([ + "deepseek-v4-pro", + "deepseek-v4-pro", + "deepseek-v4-pro", + ]); + for (const e of scoreboard.evaluations) { + expect(e.provider).toBe("deepseek"); + expect(e.summary_title.length).toBeGreaterThan(0); + expect(e.summary.toLowerCase()).toContain("example"); + expect(e.cases).toHaveLength(2); + for (const c of e.cases) { + expect(c.runs).toHaveLength(2); + for (const r of c.runs) { + expect(r.session_id).toMatch(/^session-\d{4}-\d{2}-\d{2}-\d{2}-\d{2}-\d{2}-[0-9a-f]{8}$/); + } + } + } + }); + + it("scoreboard numbers are self-consistent (case = avg of runs, evaluation = sum of cases)", async () => { + const { evaluations } = buildExampleScoreboard(); + const avg = (vals: number[]): number => vals.reduce((a, b) => a + b, 0) / vals.length; + const sum = (vals: number[]): number => vals.reduce((a, b) => a + b, 0); + for (const e of evaluations) { + for (const c of e.cases) { + expect(c.score).toBeCloseTo(avg(c.runs.map((r) => r.score)), 6); + expect(c.cost).toBeCloseTo(avg(c.runs.map((r) => r.cost)), 6); + expect(c.duration_ms).toBeCloseTo(avg(c.runs.map((r) => r.duration_ms)), 6); + } + expect(e.score).toBeCloseTo(sum(e.cases.map((c) => c.score)), 6); + expect(e.cost).toBeCloseTo(sum(e.cases.map((c) => c.cost)), 6); + expect(e.duration_ms).toBeCloseTo(sum(e.cases.map((c) => c.duration_ms)), 6); + } + }); + + it("provisionProjectAgents also seeds the example benchmark for default_agent", async () => { + await provisionProjectAgents({ root: tmpRoot, projectId: "proj_x" }); + expect( + await exists( + path.join(benchmarksDir(tmpRoot, "proj_x", DEFAULT_AGENT_ID), EXAMPLE_BENCHMARK_ID), + ), + ).toBe(true); + }); + + it("does not create benchmarks for a non-default agent", async () => { + await loadOrInitAgentState({ agentId: "worker" }); + expect(await exists(benchmarksDir(tmpRoot, DEFAULT_PROJECT_ID, "worker"))).toBe(false); + }); + + it("skips provisioning when benchmarks/ already exists (idempotent, no clobber)", async () => { + const dir = benchmarksDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID); + await fs.mkdir(dir, { recursive: true }); + await loadOrInitAgentState(); + expect(await fs.readdir(dir)).toEqual([]); + }); +}); diff --git a/packages/core/test/exec-session.test.ts b/packages/core/test/exec-session.test.ts new file mode 100644 index 0000000..bf379e6 --- /dev/null +++ b/packages/core/test/exec-session.test.ts @@ -0,0 +1,292 @@ +/** + * Behavior tests for long-running command sessions (exec_command yield + input_command). + */ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { Environment, ManagedSession } from "../src/environment/index.js"; +import { toolCall } from "../src/omnimessage/index.js"; +import type { OmniMessage } from "../src/omnimessage/index.js"; +import type { ToolConfig, ToolDefinitionConfig } from "../src/interfaces.js"; + +function execTool(overrides: Partial = {}): ToolDefinitionConfig { + return { + name: "exec_command", + description: "Run a shell command.", + parameters: { + type: "object", + properties: { + cmd: { type: "string" }, + workdir: { type: "string" }, + yield_time_ms: { type: "number" }, + }, + required: ["cmd"], + }, + permission: "rw", + maxOutputLength: 16000, + ...overrides, + }; +} + +function inputCommandTool(overrides: Partial = {}): ToolDefinitionConfig { + return { + name: "input_command", + description: "Interact with a running command session.", + parameters: { + type: "object", + properties: { + process_id: { type: "string" }, + chars: { type: "string" }, + yield_time_ms: { type: "number" }, + }, + required: ["process_id"], + }, + permission: "rw", + maxOutputLength: 16000, + ...overrides, + }; +} + +function sessionConfig(): ToolConfig { + return { customTools: [execTool(), inputCommandTool()], mcpServers: [] }; +} + +interface FinalOutput { + output: string; + stopReason?: string; +} + +/** Runs one tool call and returns the final tool_call_output's content and stop_reason. */ +async function runTool( + env: Environment, + name: string, + args: Record, + signal?: AbortSignal, +): Promise { + let last: OmniMessage | null = null; + for await (const msg of env.executeTool({ + toolCall: toolCall({ name, arguments: JSON.stringify(args), toolCallId: `call_${name}` }), + ...(signal ? { signal } : {}), + })) { + if ((msg.payload as { type?: string }).type === "tool_call_output") last = msg; + } + const p = (last?.payload ?? {}) as { output?: string; stop_reason?: string }; + return { output: p.output ?? "", stopReason: p.stop_reason }; +} + +function extractProcessId(output: string): string { + const m = output.match(/process_id (proc-[0-9a-f]+)/); + expect(m, `expected a process_id in: ${JSON.stringify(output)}`).toBeTruthy(); + return m![1]!; +} + +let tmp: string; +let env: Environment; + +beforeEach(async () => { + tmp = await mkdtemp(path.join(tmpdir(), "penguin-exec-session-")); + env = new Environment({ workspaceDir: tmp, toolConfig: sessionConfig() }); +}); + +afterEach(async () => { + env.dispose(); + await rm(tmp, { recursive: true, force: true }); +}); + +describe("exec_command — long-running command sessions", () => { + it("returns promptly when a command backgrounds a long-lived child", async () => { + // node stays resident in the background and inherits the pipes; bash exits immediately + // after the foreground echo. The old implementation waited for close (pipe EOF) -> stuck + // for 5s; the new implementation goes by the foreground exit + a short drain, returning + // within seconds, and reaps the leftover background process. + const startedAt = Date.now(); + const res = await runTool(env, "exec_command", { + cmd: 'node -e "setTimeout(()=>{},5000)" & echo hello', + }); + const elapsed = Date.now() - startedAt; + expect(elapsed).toBeLessThan(2000); + expect(res.output).toContain("hello"); + expect(res.output).not.toContain("process running with process_id"); + expect(res.stopReason).toBe("completed"); + }); + + it("streams output incrementally while the command is running", async () => { + // Two output chunks arrive 400ms apart: they should be produced as separate delta segments, + // not returned all at once when the window ends. + const deltas: string[] = []; + for await (const msg of env.executeTool({ + toolCall: toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "echo first; sleep 0.4; echo second" }), + toolCallId: "call_stream", + }), + })) { + const p = msg.payload as { type?: string; event_type?: string; output?: string }; + if (p.type === "partial_tool_call_output" && p.event_type === "delta" && p.output) { + deltas.push(p.output); + } + } + const firstIdx = deltas.findIndex((d) => d.includes("first")); + expect(firstIdx).toBeGreaterThanOrEqual(0); + expect(deltas[firstIdx]).not.toContain("second"); // first arrives earlier, not in the same segment as second + expect(deltas.slice(firstIdx + 1).some((d) => d.includes("second"))).toBe(true); + }); + + it("yields a process_id when the command is still running past yield_time_ms", async () => { + const res = await runTool(env, "exec_command", { + cmd: "sleep 30", + yield_time_ms: 300, + }); + expect(res.stopReason).toBe("completed"); + expect(res.output).toContain("process running with process_id proc-"); + }); + + it("input_command drives a running session: write stdin, get output and exit status", async () => { + const start = await runTool(env, "exec_command", { + cmd: "read line; echo got:$line", + yield_time_ms: 300, + }); + const pid = extractProcessId(start.output); + + const res = await runTool(env, "input_command", { + process_id: pid, + chars: "penguin\n", + yield_time_ms: 2000, + }); + expect(res.output).toContain("got:penguin"); + expect(res.stopReason).toBe("completed"); + }); + + it("input_command with an empty chars polls new output without writing", async () => { + const start = await runTool(env, "exec_command", { + cmd: "for i in 1 2 3; do echo line$i; sleep 0.2; done", + yield_time_ms: 100, + }); + const pid = extractProcessId(start.output); + + const res = await runTool(env, "input_command", { + process_id: pid, + chars: "", + yield_time_ms: 2000, + }); + // The command finishes during polling, yielding the remaining output and exit status. + expect(res.output).toContain("line3"); + expect(res.stopReason).toBe("completed"); + }); + + it("input_command sends Ctrl-C (U+0003) to interrupt a running session", async () => { + const start = await runTool(env, "exec_command", { + cmd: "sleep 30", + yield_time_ms: 300, + }); + const pid = extractProcessId(start.output); + + const startedAt = Date.now(); + const res = await runTool(env, "input_command", { + process_id: pid, + chars: String.fromCharCode(3), // U+0003 = Ctrl-C + yield_time_ms: 2000, + }); + const elapsed = Date.now() - startedAt; + expect(elapsed).toBeLessThan(3000); // Did not wait for the full sleep 30 + expect(res.output).not.toContain("still running"); + expect(res.stopReason).toBe("failed"); // Interrupted by signal -> non-zero exit + }); + + it("input_command rejects chars mixing U+0003 with other content", async () => { + const start = await runTool(env, "exec_command", { + cmd: "sleep 30", + yield_time_ms: 300, + }); + const pid = extractProcessId(start.output); + + const res = await runTool(env, "input_command", { + process_id: pid, + chars: `q${String.fromCharCode(3)}`, // Mixed with other content: errors, neither writes nor sends the signal + yield_time_ms: 2000, + }); + expect(res.output).toContain('send "\\u0003" alone'); + expect(res.stopReason).toBe("failed"); + + // The session was not mistakenly killed: still running. + const poll = await runTool(env, "input_command", { process_id: pid, yield_time_ms: 300 }); + expect(poll.output).toContain("still running"); + }); + + it("input_command reports an unknown process_id without throwing", async () => { + const res = await runTool(env, "input_command", { process_id: "proc-deadbeef" }); + expect(res.output).toContain("unknown process_id proc-deadbeef"); + expect(res.stopReason).toBe("failed"); + }); + + it("input_command ignores writes to a closed stdin pipe without crashing", async () => { + const start = await runTool(env, "exec_command", { + cmd: "exec 0<&-; sleep 30", + yield_time_ms: 300, + }); + const pid = extractProcessId(start.output); + + const res = await runTool(env, "input_command", { + process_id: pid, + chars: "ignored\n", + yield_time_ms: 300, + }); + expect(res.output).toContain(`process still running with process_id ${pid}`); + expect(res.stopReason).toBe("completed"); + }); + + it("runs commands through pipes, not a TTY (isTTY=false)", async () => { + const res = await runTool(env, "exec_command", { + cmd: 'node -e "process.stdout.write(String(Boolean(process.stdout.isTTY)))"', + yield_time_ms: 3000, + }); + expect(res.output).toContain("false"); + expect(res.stopReason).toBe("completed"); + }); + + it("hardens the child env against interactive hangs (editor/credentials/pager)", async () => { + const res = await runTool(env, "exec_command", { + cmd: 'echo "$GIT_EDITOR|$GIT_TERMINAL_PROMPT|$PAGER|$TERM"', + yield_time_ms: 3000, + }); + expect(res.output).toContain("true|0|cat|dumb"); + expect(res.stopReason).toBe("completed"); + }); + + it("does not start new command sessions after the environment is disposed", async () => { + env.dispose(); + const res = await runTool(env, "exec_command", { + cmd: "echo should-not-run", + yield_time_ms: 3000, + }); + expect(res.output).toContain("command session manager disposed"); + expect(res.output).not.toContain("should-not-run"); + expect(res.stopReason).toBe("failed"); + }); + + it("delivers output arriving while the consumer is suspended without waiting out the window", async () => { + // Wake-race regression: when data arrives while suspended at `yield`, its wakeup happens + // before the next wait begins (so it would be missed). collect must re-check the buffer + // right before sleeping, otherwise this batch of data would not be produced until the + // window ends (here, 5s). + const session = new ManagedSession({ cmd: "echo first; cat", cwd: tmp, env: process.env }); + try { + const gen = session.collect(5000); + const first = await gen.next(); + expect(first.done).toBe(false); + expect(String(first.value)).toContain("first"); + // The generator is still suspended at the yield above: writing to stdin now, with cat + // echoing it back, means both the data event and the wakeup have already happened. + session.write("second\n"); + await new Promise((r) => setTimeout(r, 300)); + const startedAt = Date.now(); + const next = await gen.next(); + expect(String(next.value)).toContain("second"); + expect(Date.now() - startedAt).toBeLessThan(1500); + await gen.return(undefined); + } finally { + session.kill(); + } + }); +}); diff --git a/packages/core/test/input-images.test.ts b/packages/core/test/input-images.test.ts new file mode 100644 index 0000000..4f2494f --- /dev/null +++ b/packages/core/test/input-images.test.ts @@ -0,0 +1,82 @@ +/** + * imagesToScratchpadPaths unit tests: input conversion when the session model does not support + * images -- data URL images are written to the session scratchpad and their paths appended to the + * user text; http(s) URLs are referenced as-is; image messages are removed from the input; + * image-free input is returned unchanged; images that fail to parse are replaced with an explanatory line. + */ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { mkdtemp, readFile, readdir, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { imagesToScratchpadPaths } from "../src/internal/session-support.js"; +import { imageUrlMessage, userText } from "../src/omnimessage/index.js"; +import type { TextPayload } from "../src/omnimessage/index.js"; + +const PNG_1X1 = Buffer.from( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==", + "base64", +); +const DATA_URL = `data:image/png;base64,${PNG_1X1.toString("base64")}`; + +let tmp: string; + +beforeEach(async () => { + tmp = await mkdtemp(path.join(tmpdir(), "penguin-inputimg-")); +}); + +afterEach(async () => { + await rm(tmp, { recursive: true, force: true }); +}); + +describe("imagesToScratchpadPaths", () => { + it("data URL 图片落盘、路径拼接进用户文本,图片消息移除", async () => { + const dir = path.join(tmp, "scratch", "session-1"); // auto-created if the directory doesn't exist + const out = await imagesToScratchpadPaths( + [userText("看看这两张图"), imageUrlMessage(DATA_URL), imageUrlMessage(DATA_URL)], + dir, + ); + + expect(out).toHaveLength(1); + const p = out[0]!.payload as TextPayload; + expect(p.type).toBe("text"); + expect(p.role).toBe("user"); + expect(p.text.startsWith("看看这两张图\n\n")).toBe(true); + const paths = [...p.text.matchAll(/\[attached image: ([^\]]+)\]/g)].map((m) => m[1]!); + expect(paths).toHaveLength(2); + + // The saved content matches the original image; filename = upload-<8-char random hex>.. + const files = await readdir(dir); + expect(files).toHaveLength(2); + for (const f of paths) { + expect(path.dirname(f)).toBe(dir); + expect(path.basename(f)).toMatch(/^upload-[0-9a-f]{8}\.png$/); + expect(await readFile(f)).toEqual(PNG_1X1); + } + // The two images' random names differ from each other. + expect(new Set(paths).size).toBe(2); + }); + + it("http(s) URL 不落盘,原样引用;仅图片输入时补一条纯路径文本", async () => { + const out = await imagesToScratchpadPaths([imageUrlMessage("https://example.com/a.png")], tmp); + expect(out).toHaveLength(1); + const p = out[0]!.payload as TextPayload; + expect(p.type).toBe("text"); + expect(p.text).toBe("[attached image: https://example.com/a.png]"); + expect(await readdir(tmp)).toHaveLength(0); + }); + + it("无图片输入原样返回(不触碰文件系统)", async () => { + const input = [userText("纯文本")]; + const out = await imagesToScratchpadPaths(input, path.join(tmp, "untouched")); + expect(out).toBe(input); + }); + + it("无法解析的图片以说明行代替,不静默丢失", async () => { + const out = await imagesToScratchpadPaths( + [userText("hi"), imageUrlMessage("data:text/plain,oops")], + tmp, + ); + const p = out[0]!.payload as TextPayload; + expect(p.text).toContain("could not be saved"); + }); +}); diff --git a/packages/core/test/llm.e2e.test.ts b/packages/core/test/llm.e2e.test.ts new file mode 100644 index 0000000..50be7f2 --- /dev/null +++ b/packages/core/test/llm.e2e.test.ts @@ -0,0 +1,150 @@ +/** + * GenerativeModel live e2e. The whole suite is it.skip by default and stays offline; + * requires an explicit opt-in to run against a real endpoint (this avoids ordinary unit + * test runs firing real network requests just because an API key happens to be present): + * PENGUIN_E2E=1 pnpm test # or pnpm test:e2e + * The key comes from .env (gitignored) or an environment variable; the provider is picked + * by whichever key is available (CI uses DeepSeek). + */ +import "dotenv/config"; +import { describe, expect, it } from "vitest"; + +import { GenerativeModel } from "../src/llm/index.js"; +import { toolCallOutput, userText } from "../src/omnimessage/index.js"; +import type { OmniMessage, ToolCallPayload } from "../src/omnimessage/index.js"; + +/** Picks a provider in order by available key; AgentHub routes by modelId and auto-reads the matching env var. */ +const PROVIDERS = [ + { key: "ANTHROPIC_API_KEY", modelId: "claude-sonnet-4-6" }, + { key: "DEEPSEEK_API_KEY", modelId: "deepseek-v4-flash" }, +] as const; +const provider = PROVIDERS.find((p) => process.env[p.key]); +const runLive = process.env.PENGUIN_E2E === "1" && provider !== undefined; +const maybe = runLive ? it : it.skip; + +describe(`GenerativeModel live e2e (${provider?.modelId ?? "skipped"})`, () => { + maybe( + "streams a short reply with partial text, a complete text, and token usage", + async () => { + const model = new GenerativeModel({ + modelId: provider!.modelId, + tools: [], + systemPrompt: "You are concise.", + // Generous budget: a reasoning model's thinking may take up a lot of space, leave room for the reply to land reliably. + maxTokens: 2048, + }); + + const out: OmniMessage[] = []; + for await (const msg of model.streamGenerate({ + newMessages: [userText("Say hello in exactly three words.")], + })) { + out.push(msg); + } + + const payloadType = (m: OmniMessage): string => (m.payload as { type: string }).type; + + // At least 1 partial_text. + const partialTexts = out.filter((m) => payloadType(m) === "partial_text"); + expect(partialTexts.length).toBeGreaterThanOrEqual(1); + + // At least 1 non-empty complete text. + const completeTexts = out.filter((m) => payloadType(m) === "text"); + expect(completeTexts.length).toBeGreaterThanOrEqual(1); + const text = (completeTexts[0]!.payload as { text: string }).text; + expect(text.trim().length).toBeGreaterThan(0); + + // Trailing token_usage, with request.total > 0. + const last = out.at(-1)!; + expect(payloadType(last)).toBe("token_usage"); + const usage = last.payload as { request: { total: number } }; + expect(usage.request.total).toBeGreaterThan(0); + + // Session-cumulative tokens are also recorded on the instance. + expect(model.sessionTokens.total).toBeGreaterThan(0); + }, + // Live calls have inherent network/server jitter: allow a generous timeout and retries so + // one slow API call doesn't fail CI (assertions stay strict; only transient timeouts are tolerated). + { timeout: 90_000, retry: 2 }, + ); +}); + +// --- Gemini tool_call_id uniqueness live regression (consecutive same-name tool calls made the +// frontend tool cards overwrite each other) --- +// Gemini's functionCall has no call id, so AgentHub uses the function name as tool_call_id; this group +// verifies EventTranslator's in-Session uniqueness (#n suffix) and outbound restoration +// (functionResponse paired by function name) round-trip on the real API. Runs only when explicitly +// opted in with GEMINI_API_KEY set. +const runGemini = process.env.PENGUIN_E2E === "1" && process.env.GEMINI_API_KEY !== undefined; +const maybeGemini = runGemini ? it : it.skip; + +describe(`GenerativeModel live e2e (gemini-3.5-flash: tool_call_id 唯一化${runGemini ? "" : ",skipped"})`, () => { + maybeGemini( + "连续两轮同名工具调用拿到不同 tool_call_id;带后缀 id 的 tool_result 回传被正确配对", + async () => { + const model = new GenerativeModel({ + modelId: "gemini-3.5-flash", + tools: [ + { + name: "get_time", + description: "Get the current local time of a city.", + parameters: { + type: "object", + properties: { city: { type: "string", description: "City name" } }, + required: ["city"], + }, + }, + ], + systemPrompt: + "You are a tool-driven assistant. Always use the get_time tool to answer time questions, one call per turn. Never guess.", + maxTokens: 1024, + thinkingLevel: "none", + }); + + const round = async ( + input: OmniMessage[], + ): Promise<{ calls: ToolCallPayload[]; text: string }> => { + const calls: ToolCallPayload[] = []; + let text = ""; + const gen = model.streamGenerate({ newMessages: input }); + let res = await gen.next(); + while (!res.done) { + const p = res.value.payload as { type: string }; + if (p.type === "tool_call") calls.push(res.value.payload as ToolCallPayload); + if (p.type === "text") text += (res.value.payload as { text: string }).text; + res = await gen.next(); + } + expect((res.value as { status: string }).status).toBe("completed"); + return { calls, text }; + }; + + // Round 1: call get_time(Tokyo); the id keeps the function name. + const r1 = await round([ + userText( + "What time is it in Tokyo? After you get that result, also check Paris (one tool call at a time).", + ), + ]); + expect(r1.calls.length).toBeGreaterThanOrEqual(1); + expect(r1.calls[0]!.tool_call_id).toBe("get_time"); + + // Round 2: after returning the result the model checks Paris next — the same-name call must get a new suffixed id. + const r2 = await round( + r1.calls.map((c) => + toolCallOutput({ output: "10:00 AM (mock)", toolCallId: c.tool_call_id }), + ), + ); + expect(r2.calls.length).toBeGreaterThanOrEqual(1); + expect(r2.calls[0]!.tool_call_id).toBe("get_time#2"); + + // Round 3: after returning the tool_result carrying the #2 suffix (outbound strips it back to the + // function name), the provider finishes normally. Reaching completed means the functionResponse + // pairing was accepted; some models may keep appending calls, so don't hard-assert the body. + const r3 = await round( + r2.calls.map((c) => + toolCallOutput({ output: "3:00 AM (mock)", toolCallId: c.tool_call_id }), + ), + ); + expect(r3.calls.length + r3.text.length).toBeGreaterThan(0); + }, + { timeout: 120_000, retry: 2 }, + ); +}); diff --git a/packages/core/test/llm.test.ts b/packages/core/test/llm.test.ts new file mode 100644 index 0000000..06f3682 --- /dev/null +++ b/packages/core/test/llm.test.ts @@ -0,0 +1,1656 @@ +/** + * GenerativeModel pure unit tests (no network). + * + * Covers two core pieces of logic: + * 1. Merging OmniMessage[] into one UniMessage (including throwing on mixed roles, and + * mapping each content type); + * 2. Translating/aggregating UniEvent[] into OmniMessage[] (partial_* ordering, complete + * messages, token_usage accumulation, tool_call_id passthrough). + * As well as helper functions for token conversion, UniConfig construction, and retry + * determination. + */ +import { describe, expect, it } from "vitest"; +import { ThinkingLevel } from "@prismshadow/agenthub"; +import type { UniEvent, UniMessage, UsageMetadata } from "@prismshadow/agenthub"; +import type { LLMOutcome } from "../src/interfaces.js"; + +import { + EventTranslator, + GenerativeModel, + ToolCallIdAllocator, + buildUniConfig, + isIncompleteStreamError, + isMalformedJsonParseError, + isRetryableError, + mapThinkingLevel, + mergeOmniToUniMessage, + stripToolCallIdSuffix, + toolDefinitionsToSchemas, + translateEvents, + usageToTokenCounts, +} from "../src/llm/index.js"; +import { + assistantText, + imageUrlMessage, + inlineData, + inlineThinking, + thinkingMessage, + toolCall, + toolCallOutput, + userText, +} from "../src/omnimessage/index.js"; +import type { + OmniMessage, + TextPayload, + ThinkingPayload, + ToolCallPayload, + TokenUsagePayload, +} from "../src/omnimessage/index.js"; + +// Small helper to construct a UniEvent. +function ev(partial: Partial & Pick): UniEvent { + return { + role: "assistant", + event_type: "delta", + usage_metadata: null, + finish_reason: null, + ...partial, + }; +} + +describe("mergeOmniToUniMessage", () => { + it("merges same-role messages into one UniMessage and maps content types", () => { + const uni = mergeOmniToUniMessage([ + userText("hello"), + imageUrlMessage("https://example.com/a.png"), + inlineData("user", Buffer.from("xyz").toString("base64"), "image/png"), + ]); + expect(uni.role).toBe("user"); + expect(uni.content_items).toHaveLength(3); + expect(uni.content_items[0]).toEqual({ type: "text", text: "hello" }); + expect(uni.content_items[1]).toEqual({ + type: "image_url", + image_url: "https://example.com/a.png", + }); + const inline = uni.content_items[2]!; + expect(inline.type).toBe("inline_data"); + if (inline.type === "inline_data") { + expect(inline.mime_type).toBe("image/png"); + expect(Buffer.isBuffer(inline.data)).toBe(true); + expect(inline.data.toString()).toBe("xyz"); + } + }); + + it("maps assistant thinking and inline_thinking content", () => { + const uni = mergeOmniToUniMessage([ + thinkingMessage("step by step"), + inlineThinking(Buffer.from("sig").toString("base64"), "application/octet-stream"), + ]); + expect(uni.role).toBe("assistant"); + expect(uni.content_items).toHaveLength(2); + expect(uni.content_items[0]).toEqual({ + type: "thinking", + thinking: "step by step", + }); + const inline = uni.content_items[1]!; + expect(inline.type).toBe("inline_thinking"); + if (inline.type === "inline_thinking") { + expect(inline.mime_type).toBe("application/octet-stream"); + expect(Buffer.isBuffer(inline.data)).toBe(true); + expect(inline.data.toString()).toBe("sig"); + } + }); + + it("maps assistant tool_call OmniMessage (args JSON string → object)", () => { + const uni = mergeOmniToUniMessage([ + toolCall({ + name: "exec_command", + arguments: '{"cmd":"ls -la"}', + toolCallId: "call_1", + }), + ]); + expect(uni.role).toBe("assistant"); + const item = uni.content_items[0]!; + expect(item).toEqual({ + type: "tool_call", + name: "exec_command", + arguments: { cmd: "ls -la" }, + tool_call_id: "call_1", + }); + }); + + it("maps tool_call_output to tool_result with role user and preserves id", () => { + const uni = mergeOmniToUniMessage([ + toolCallOutput({ output: "total 0", toolCallId: "call_1" }), + ]); + expect(uni.role).toBe("user"); + expect(uni.content_items[0]).toEqual({ + type: "tool_result", + text: "total 0", + tool_call_id: "call_1", + }); + }); + + it("maps tool_call_output images to tool_result.images (data URL array)", () => { + const dataUrl = "data:image/png;base64,AAAA"; + const uni = mergeOmniToUniMessage([ + toolCallOutput({ output: "image/png, 4 B", toolCallId: "call_img", images: [dataUrl] }), + ]); + expect(uni.role).toBe("user"); + expect(uni.content_items[0]).toEqual({ + type: "tool_result", + text: "image/png, 4 B", + images: [dataUrl], + tool_call_id: "call_img", + }); + }); + + it("throws on mixed roles", () => { + expect(() => mergeOmniToUniMessage([userText("hi"), thinkingMessage("reasoning")])).toThrow( + /mixed roles/, + ); + }); + + it("throws on empty input", () => { + expect(() => mergeOmniToUniMessage([])).toThrow(); + }); +}); + +describe("usageToTokenCounts", () => { + it("maps cached→cache_read, prompt→cache_write, thoughts+response→output", () => { + const usage: UsageMetadata = { + cached_tokens: 5, + prompt_tokens: 10, + thoughts_tokens: 3, + response_tokens: 7, + }; + // cache_read = 5; cache_write = 10 (non-cached input); output = 3 + 7 = 10; total = 25. + expect(usageToTokenCounts(usage)).toEqual({ + cache_read: 5, + cache_write: 10, + output: 10, + total: 25, + }); + }); + + it("treats nulls as zero", () => { + const usage: UsageMetadata = { + cached_tokens: null, + prompt_tokens: null, + thoughts_tokens: null, + response_tokens: null, + }; + expect(usageToTokenCounts(usage)).toEqual({ + cache_read: 0, + cache_write: 0, + output: 0, + total: 0, + }); + }); +}); + +describe("translateEvents", () => { + it("emits text partials (start/delta/stop), a complete text, and token_usage", () => { + const events: UniEvent[] = [ + ev({ event_type: "start", content_items: [] }), + ev({ content_items: [{ type: "text", text: "Hel" }] }), + ev({ content_items: [{ type: "text", text: "lo" }] }), + ev({ + event_type: "stop", + content_items: [], + finish_reason: "stop", + usage_metadata: { + cached_tokens: 0, + prompt_tokens: 12, + thoughts_tokens: 0, + response_tokens: 4, + }, + }), + ]; + const { messages, requestTokens, sessionTokens } = translateEvents(events); + + const types = messages.map((m) => (m.payload as { type: string }).type); + // partial start, two deltas, partial stop, complete text, token_usage. + expect(types).toEqual([ + "partial_text", + "partial_text", + "partial_text", + "partial_text", + "text", + "token_usage", + ]); + + // partial events: start (empty) → delta "Hel" → delta "lo" → stop. + const ptexts = messages + .filter((m) => (m.payload as { type: string }).type === "partial_text") + .map((m) => m.payload as { event_type: string; text: string }); + expect(ptexts).toEqual([ + { + type: "partial_text", + role: "assistant", + event_type: "start", + text: "", + stop_reason: "completed", + }, + { + type: "partial_text", + role: "assistant", + event_type: "delta", + text: "Hel", + stop_reason: "completed", + }, + { + type: "partial_text", + role: "assistant", + event_type: "delta", + text: "lo", + stop_reason: "completed", + }, + { + type: "partial_text", + role: "assistant", + event_type: "stop", + text: "", + stop_reason: "completed", + }, + ]); + + // complete text message: concatenated, stop_reason completed (finish_reason "stop"). + const complete = messages.find((m) => (m.payload as { type: string }).type === "text")! + .payload as TextPayload; + expect(complete.text).toBe("Hello"); + expect(complete.role).toBe("assistant"); + expect(complete.stop_reason).toBe("completed"); + + // token accounting: request total = 12 + 4 = 16. + expect(requestTokens.total).toBe(16); + expect(requestTokens.output).toBe(4); + expect(sessionTokens).toEqual(requestTokens); + + const tu = messages.at(-1)!.payload as TokenUsagePayload; + expect(tu.type).toBe("token_usage"); + expect(tu.request.total).toBe(16); + expect(tu.session.total).toBe(16); + }); + + it("accumulates partial_tool_call args, uses complete tool_call as authoritative, preserves id", () => { + const events: UniEvent[] = [ + ev({ + event_type: "start", + content_items: [ + { type: "partial_tool_call", name: "exec_command", arguments: "", tool_call_id: "c1" }, + ], + }), + ev({ + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"cmd":"ls', + tool_call_id: "c1", + }, + ], + }), + ev({ + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: ' -la"}', + tool_call_id: "c1", + }, + ], + }), + ev({ + event_type: "stop", + finish_reason: "tool_call", + content_items: [ + { + type: "tool_call", + name: "exec_command", + arguments: { cmd: "ls -la" }, + tool_call_id: "c1", + }, + ], + }), + ]; + const { messages } = translateEvents(events); + const types = messages.map((m) => (m.payload as { type: string }).type); + expect(types).toEqual([ + "partial_tool_call", // start + "partial_tool_call", // delta + "partial_tool_call", // delta + "partial_tool_call", // stop + "tool_call", // complete + "token_usage", + ]); + + // partial start carries name, no args; deltas carry arg fragments. + const partials = messages + .filter((m) => (m.payload as { type: string }).type === "partial_tool_call") + .map((m) => m.payload as { event_type: string; arguments: string; tool_call_id: string }); + expect(partials[0]!.event_type).toBe("start"); + expect(partials[0]!.arguments).toBe(""); + expect(partials[1]!.arguments).toBe('{"cmd":"ls'); + expect(partials[2]!.arguments).toBe(' -la"}'); + expect(partials[3]!.event_type).toBe("stop"); + expect(partials.every((p) => p.tool_call_id === "c1")).toBe(true); + + // complete tool_call uses the authoritative complete content item. + const tc = messages.find((m) => (m.payload as { type: string }).type === "tool_call")! + .payload as ToolCallPayload; + expect(tc.name).toBe("exec_command"); + expect(tc.tool_call_id).toBe("c1"); + expect(tc.arguments).toBe('{"cmd":"ls -la"}'); + expect(tc.stop_reason).toBe("completed"); + }); + + it("falls back to accumulated arg buffer when no complete tool_call item arrives", () => { + const events: UniEvent[] = [ + ev({ + event_type: "start", + content_items: [ + { type: "partial_tool_call", name: "do_it", arguments: '{"x":', tool_call_id: "z9" }, + ], + }), + ev({ + content_items: [ + { type: "partial_tool_call", name: "do_it", arguments: "1}", tool_call_id: "z9" }, + ], + }), + ev({ event_type: "stop", finish_reason: "tool_call", content_items: [] }), + ]; + const { messages } = translateEvents(events); + const tc = messages.find((m) => (m.payload as { type: string }).type === "tool_call")! + .payload as ToolCallPayload; + expect(tc.arguments).toBe('{"x":1}'); + expect(tc.tool_call_id).toBe("z9"); + }); + + it("ignores tool-call fragments with empty tool_call_id (no spurious empty tool_call)", () => { + // Regression: some early streamed fragments may carry an empty tool_call_id; this must not + // be used to generate an empty tool_call (otherwise it would trigger "Unknown tool" and + // AgentHub's "tool_call_id is required" error). + const events: UniEvent[] = [ + ev({ + event_type: "start", + content_items: [ + { type: "partial_tool_call", name: "exec_command", arguments: "", tool_call_id: "real1" }, + ], + }), + // A streamed fragment with an empty id mixed in. + ev({ + content_items: [{ type: "partial_tool_call", name: "", arguments: "", tool_call_id: "" }], + }), + ev({ + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"cmd":"ls"}', + tool_call_id: "real1", + }, + ], + }), + ev({ + event_type: "stop", + finish_reason: "tool_call", + content_items: [ + { + type: "tool_call", + name: "exec_command", + arguments: { cmd: "ls" }, + tool_call_id: "real1", + }, + // A complete tool_call with an empty id should also be ignored. + { type: "tool_call", name: "", arguments: {}, tool_call_id: "" }, + ], + }), + ]; + const { messages } = translateEvents(events); + const toolCalls = messages.filter((m) => (m.payload as { type: string }).type === "tool_call"); + expect(toolCalls).toHaveLength(1); + expect((toolCalls[0]!.payload as ToolCallPayload).tool_call_id).toBe("real1"); + // No message should carry an empty tool_call_id. + const emptyIds = messages.filter( + (m) => (m.payload as { tool_call_id?: string }).tool_call_id === "", + ); + expect(emptyIds).toHaveLength(0); + }); + + it("attributes empty-id tool-call argument deltas to the active tool call", () => { + const events: UniEvent[] = [ + ev({ + event_type: "start", + content_items: [ + { type: "partial_tool_call", name: "exec_command", arguments: "", tool_call_id: "real1" }, + ], + }), + ev({ + content_items: [ + { type: "partial_tool_call", name: "", arguments: '{"cmd":"l', tool_call_id: "" }, + ], + }), + ev({ + content_items: [ + { type: "partial_tool_call", name: "", arguments: 's"}', tool_call_id: "" }, + ], + }), + ev({ + event_type: "stop", + finish_reason: "tool_call", + content_items: [ + { + type: "tool_call", + name: "exec_command", + arguments: { cmd: "ls" }, + tool_call_id: "real1", + }, + ], + }), + ]; + const { messages } = translateEvents(events); + const deltas = messages + .filter( + (m) => + (m.payload as { type: string }).type === "partial_tool_call" && + (m.payload as { event_type: string }).event_type === "delta", + ) + .map((m) => m.payload as { arguments: string; tool_call_id: string }); + + expect(deltas.map((p) => p.arguments)).toEqual(['{"cmd":"l', 's"}']); + expect(deltas.every((p) => p.tool_call_id === "real1")).toBe(true); + }); + + it("emits a complete tool_call immediately when its complete content item arrives mid-stream (async/incremental)", () => { + // Two tools: the first's complete content item arrives mid-stream (not at finish) -> should + // be produced immediately. + const events: UniEvent[] = [ + ev({ + event_type: "start", + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"cmd":"a"}', + tool_call_id: "t1", + }, + ], + }), + // t1's complete content item arrives early (before t2), and should finish t1 immediately. + ev({ + content_items: [ + { type: "tool_call", name: "exec_command", arguments: { cmd: "a" }, tool_call_id: "t1" }, + ], + }), + ev({ + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"cmd":"b"}', + tool_call_id: "t2", + }, + { type: "tool_call", name: "exec_command", arguments: { cmd: "b" }, tool_call_id: "t2" }, + ], + }), + ev({ event_type: "stop", finish_reason: "tool_call", content_items: [] }), + ]; + const { messages } = translateEvents(events); + const completeToolCalls = messages.filter( + (m) => (m.payload as { type: string }).type === "tool_call", + ); + // Both complete tool_calls are produced, with t1 before t2 (in arrival order, not as a + // batch at finish). + expect(completeToolCalls.map((m) => (m.payload as ToolCallPayload).tool_call_id)).toEqual([ + "t1", + "t2", + ]); + // t1's complete tool_call appears before t2's start fragment (proving it was produced + // before finish). + const idxT1Complete = messages.findIndex( + (m) => + (m.payload as { type: string }).type === "tool_call" && + (m.payload as ToolCallPayload).tool_call_id === "t1", + ); + const idxT2Start = messages.findIndex( + (m) => + (m.payload as { type: string }).type === "partial_tool_call" && + (m.payload as { tool_call_id?: string }).tool_call_id === "t2", + ); + expect(idxT1Complete).toBeLessThan(idxT2Start); + // Each tool is produced exactly once (no duplication at finish). + expect(completeToolCalls).toHaveLength(2); + }); + + it("does not write name on delta or stop tool-call partials", () => { + const events: UniEvent[] = [ + ev({ + event_type: "start", + content_items: [ + { type: "partial_tool_call", name: "exec_command", arguments: "", tool_call_id: "c1" }, + ], + }), + ev({ + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"cmd":"ls"}', + tool_call_id: "c1", + }, + ], + }), + ev({ event_type: "stop", finish_reason: "tool_call", content_items: [] }), + ]; + const { messages } = translateEvents(events); + const partials = messages.filter( + (m) => (m.payload as { type: string }).type === "partial_tool_call", + ) as { payload: { event_type: string; name: string } }[]; + const start = partials.find((p) => p.payload.event_type === "start")!; + const delta = partials.find((p) => p.payload.event_type === "delta")!; + const stop = partials.find((p) => p.payload.event_type === "stop")!; + expect(start.payload.name).toBe("exec_command"); // start still carries name. + expect(delta.payload.name).toBe(""); // delta does not carry name. + expect(stop.payload.name).toBe(""); // stop does not carry name. + }); + + it("emits thinking partials and a complete thinking message before text", () => { + const events: UniEvent[] = [ + ev({ event_type: "start", content_items: [{ type: "thinking", thinking: "Let me" }] }), + ev({ content_items: [{ type: "thinking", thinking: " think" }] }), + ev({ content_items: [{ type: "text", text: "Answer" }] }), + ev({ event_type: "stop", finish_reason: "stop", content_items: [] }), + ]; + const { messages } = translateEvents(events); + const completeTypes = messages + .map((m) => (m.payload as { type: string }).type) + .filter((t) => t === "thinking" || t === "text"); + // thinking complete message emitted before text complete message. + expect(completeTypes).toEqual(["thinking", "text"]); + + const think = messages.find((m) => (m.payload as { type: string }).type === "thinking")! + .payload as ThinkingPayload; + expect(think.thinking).toBe("Let me think"); + }); + + it("emits complete thinking and text before a mid-stream complete tool_call", () => { + // Reproduces the "thinking ends up after tool_call in Trace" regression: the model thinks + // first, then outputs text, then the tool call's complete content item arrives before + // finish. The complete-message order must be thinking -> text -> tool_call (not + // tool_call -> thinking -> text). + const events: UniEvent[] = [ + ev({ event_type: "start", content_items: [{ type: "thinking", thinking: "I should" }] }), + ev({ content_items: [{ type: "thinking", thinking: " run ls" }] }), + ev({ content_items: [{ type: "text", text: "Running it." }] }), + ev({ + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"cmd":"ls"}', + tool_call_id: "c1", + }, + { type: "tool_call", name: "exec_command", arguments: { cmd: "ls" }, tool_call_id: "c1" }, + ], + }), + ev({ event_type: "stop", finish_reason: "tool_call", content_items: [] }), + ]; + const { messages } = translateEvents(events); + + const completeTypes = messages + .map((m) => (m.payload as { type: string }).type) + .filter((t) => t === "thinking" || t === "text" || t === "tool_call"); + // Complete-message order: thinking -> text -> tool_call, each exactly once (flush does not repeat). + expect(completeTypes).toEqual(["thinking", "text", "tool_call"]); + + // Complete thinking/text are marked completed when finished at the boundary (the finish + // reason belongs to the tool_call itself). + const think = messages.find((m) => (m.payload as { type: string }).type === "thinking")! + .payload as ThinkingPayload; + expect(think.thinking).toBe("I should run ls"); + expect(think.stop_reason).toBe("completed"); + const text = messages.find((m) => (m.payload as { type: string }).type === "text")! + .payload as TextPayload; + expect(text.text).toBe("Running it."); + expect(text.stop_reason).toBe("completed"); + const tc = messages.find((m) => (m.payload as { type: string }).type === "tool_call")! + .payload as ToolCallPayload; + expect(tc.stop_reason).toBe("completed"); + }); + + it("flushes thinking/text emitted after a tool_call (does not drop later segments)", () => { + // Interleaved output: text appears both before and after tool_call. The reset-after-flush + // design should let a new text segment following tool_call still be produced correctly at + // finish (a one-shot guard would lose it). + const events: UniEvent[] = [ + ev({ event_type: "start", content_items: [{ type: "text", text: "before " }] }), + ev({ content_items: [{ type: "text", text: "call" }] }), + ev({ + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"cmd":"ls"}', + tool_call_id: "c1", + }, + { type: "tool_call", name: "exec_command", arguments: { cmd: "ls" }, tool_call_id: "c1" }, + ], + }), + ev({ content_items: [{ type: "text", text: "after call" }] }), + ev({ event_type: "stop", finish_reason: "stop", content_items: [] }), + ]; + const { messages } = translateEvents(events); + + const completeTypes = messages + .map((m) => (m.payload as { type: string }).type) + .filter((t) => t === "text" || t === "tool_call"); + // Produces the text before tool_call first, then tool_call, then the new text segment after it. + expect(completeTypes).toEqual(["text", "tool_call", "text"]); + + const texts = messages + .filter((m) => (m.payload as { type: string }).type === "text") + .map((m) => (m.payload as TextPayload).text); + expect(texts).toEqual(["before call", "after call"]); + }); + + it("keeps thinking and text as separate segments across a thinking→text boundary", () => { + const events: UniEvent[] = [ + ev({ event_type: "start", content_items: [{ type: "thinking", thinking: "ponder" }] }), + ev({ content_items: [{ type: "text", text: "answer" }] }), + ev({ event_type: "stop", finish_reason: "stop", content_items: [] }), + ]; + const { messages } = translateEvents(events); + const completeTypes = messages + .map((m) => (m.payload as { type: string }).type) + .filter((t) => t === "thinking" || t === "text"); + expect(completeTypes).toEqual(["thinking", "text"]); + // The thinking segment finishes before the text segment starts: partial_thinking stop + // precedes partial_text start. + const idxThinkStop = messages.findIndex( + (m) => + (m.payload as { type: string; event_type?: string }).type === "partial_thinking" && + (m.payload as { event_type?: string }).event_type === "stop", + ); + const idxTextStart = messages.findIndex( + (m) => + (m.payload as { type: string; event_type?: string }).type === "partial_text" && + (m.payload as { event_type?: string }).event_type === "start", + ); + expect(idxThinkStop).toBeGreaterThanOrEqual(0); + expect(idxThinkStop).toBeLessThan(idxTextStart); + }); + + it("emits text before thinking for a text→thinking boundary (not reordered)", () => { + const events: UniEvent[] = [ + ev({ event_type: "start", content_items: [{ type: "text", text: "hello" }] }), + ev({ content_items: [{ type: "thinking", thinking: "hmm" }] }), + ev({ event_type: "stop", finish_reason: "stop", content_items: [] }), + ]; + const { messages } = translateEvents(events); + const completeTypes = messages + .map((m) => (m.payload as { type: string }).type) + .filter((t) => t === "thinking" || t === "text"); + // The generation order is text -> thinking, and the complete-message order must match it + // (the old implementation would reverse it to put thinking first). + expect(completeTypes).toEqual(["text", "thinking"]); + }); + + it("does not merge two thinking segments separated by text (think→text→think)", () => { + const events: UniEvent[] = [ + ev({ event_type: "start", content_items: [{ type: "thinking", thinking: "first" }] }), + ev({ content_items: [{ type: "text", text: "mid" }] }), + ev({ content_items: [{ type: "thinking", thinking: "second" }] }), + ev({ event_type: "stop", finish_reason: "stop", content_items: [] }), + ]; + const { messages } = translateEvents(events); + const completeTypes = messages + .map((m) => (m.payload as { type: string }).type) + .filter((t) => t === "thinking" || t === "text"); + // Three separate segments, produced in generation order (the old implementation merged + // the two thinking segments and had no second start). + expect(completeTypes).toEqual(["thinking", "text", "thinking"]); + const thinkings = messages + .filter((m) => (m.payload as { type: string }).type === "thinking") + .map((m) => (m.payload as ThinkingPayload).thinking); + expect(thinkings).toEqual(["first", "second"]); + // The second thinking segment reopens a segment: two partial_thinking starts appear. + const thinkStarts = messages.filter( + (m) => + (m.payload as { type: string; event_type?: string }).type === "partial_thinking" && + (m.payload as { event_type?: string }).event_type === "start", + ); + expect(thinkStarts).toHaveLength(2); + }); + + it("does not merge two text segments separated by thinking (text→think→text)", () => { + const events: UniEvent[] = [ + ev({ event_type: "start", content_items: [{ type: "text", text: "a" }] }), + ev({ content_items: [{ type: "thinking", thinking: "b" }] }), + ev({ content_items: [{ type: "text", text: "c" }] }), + ev({ event_type: "stop", finish_reason: "stop", content_items: [] }), + ]; + const { messages } = translateEvents(events); + const completeTypes = messages + .map((m) => (m.payload as { type: string }).type) + .filter((t) => t === "thinking" || t === "text"); + expect(completeTypes).toEqual(["text", "thinking", "text"]); + const texts = messages + .filter((m) => (m.payload as { type: string }).type === "text") + .map((m) => (m.payload as TextPayload).text); + expect(texts).toEqual(["a", "c"]); + }); + + it("flushes thinking/text before a partial-only tool_call (no full item until finish)", () => { + // The tool only goes through partial_tool_call deltas (no complete tool_call content item), + // and is produced by falling back at finish. The new tool's first delta is a type boundary, + // so thinking/text must be flushed first. + const events: UniEvent[] = [ + ev({ event_type: "start", content_items: [{ type: "thinking", thinking: "plan" }] }), + ev({ content_items: [{ type: "text", text: "doing" }] }), + ev({ + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"cmd":', + tool_call_id: "p1", + }, + ], + }), + ev({ + content_items: [ + { type: "partial_tool_call", name: "", arguments: '"ls"}', tool_call_id: "p1" }, + ], + }), + ev({ event_type: "stop", finish_reason: "tool_call", content_items: [] }), + ]; + const { messages } = translateEvents(events); + const completeTypes = messages + .map((m) => (m.payload as { type: string }).type) + .filter((t) => t === "thinking" || t === "text" || t === "tool_call"); + expect(completeTypes).toEqual(["thinking", "text", "tool_call"]); + // Only one thinking, one text, each flushed exactly once before the tool's first delta. + const tc = messages.find((m) => (m.payload as { type: string }).type === "tool_call")! + .payload as ToolCallPayload; + expect(tc.arguments).toBe('{"cmd":"ls"}'); + }); + + it("does not re-flush on continuation tool deltas lacking a tool_call_id", () => { + // Some providers' subsequent argument deltas do not carry an id, and are attributed to + // activeToolCallId; this must not trigger a duplicate flush, nor produce a spurious + // empty thinking/text complete message. + const events: UniEvent[] = [ + ev({ event_type: "start", content_items: [{ type: "thinking", thinking: "go" }] }), + ev({ + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"a":', + tool_call_id: "k1", + }, + ], + }), + // A continuation delta with no id. + ev({ + content_items: [{ type: "partial_tool_call", name: "", arguments: "1}", tool_call_id: "" }], + }), + ev({ event_type: "stop", finish_reason: "tool_call", content_items: [] }), + ]; + const { messages } = translateEvents(events); + const thinkings = messages.filter((m) => (m.payload as { type: string }).type === "thinking"); + const texts = messages.filter((m) => (m.payload as { type: string }).type === "text"); + expect(thinkings).toHaveLength(1); // Exactly one, not duplicated by the continuation delta + expect(texts).toHaveLength(0); // Does not conjure an empty text out of nowhere + const toolStarts = messages.filter( + (m) => + (m.payload as { type: string; event_type?: string }).type === "partial_tool_call" && + (m.payload as { event_type?: string }).event_type === "start", + ); + expect(toolStarts).toHaveLength(1); // The same tool has only one start + }); + + it("accumulates session tokens across two requests", () => { + const mkUsage = (p: number, r: number): UsageMetadata => ({ + cached_tokens: 0, + prompt_tokens: p, + thoughts_tokens: 0, + response_tokens: r, + }); + const first = translateEvents([ + ev({ content_items: [{ type: "text", text: "a" }] }), + ev({ + event_type: "stop", + finish_reason: "stop", + content_items: [], + usage_metadata: mkUsage(10, 5), + }), + ]); + expect(first.sessionTokens.total).toBe(15); + + const second = translateEvents( + [ + ev({ content_items: [{ type: "text", text: "b" }] }), + ev({ + event_type: "stop", + finish_reason: "stop", + content_items: [], + usage_metadata: mkUsage(20, 3), + }), + ], + first.sessionTokens, + ); + expect(second.requestTokens.total).toBe(23); + expect(second.sessionTokens.total).toBe(38); + }); + + it("keeps only the last usage snapshot within a request (per-chunk cumulative reports are not summed)", () => { + // Regression: Gemini (and some OpenAI-compatible endpoints) report usage as a **cumulative + // snapshot** per chunk; summing them would inflate usage by roughly the chunk count + // (especially for output), so the last snapshot must be authoritative. + const mkUsage = (p: number, t: number, r: number): UsageMetadata => ({ + cached_tokens: null, + prompt_tokens: p, + thoughts_tokens: t, + response_tokens: r, + }); + const { messages, requestTokens, sessionTokens } = translateEvents([ + ev({ content_items: [{ type: "text", text: "Hel" }], usage_metadata: mkUsage(16, 488, 18) }), + ev({ content_items: [{ type: "text", text: "lo" }], usage_metadata: mkUsage(16, 488, 20) }), + ev({ + event_type: "stop", + finish_reason: "stop", + content_items: [], + usage_metadata: mkUsage(16, 488, 20), + }), + ]); + // The last snapshot is authoritative: cache_write = 16, output = 488 + 20 = 508, total = 524. + expect(requestTokens).toEqual({ cache_read: 0, cache_write: 16, output: 508, total: 524 }); + expect(sessionTokens).toEqual(requestTokens); + const tu = messages.at(-1)!.payload as TokenUsagePayload; + expect(tu.request.total).toBe(524); + }); +}); + +describe("EventTranslator.finishInterrupted (PRN-012 结构闭合)", () => { + it("closes an open text segment with a stop + complete text marked with the interruption reason, and emits no token_usage", () => { + const tr = new EventTranslator(); + const out: OmniMessage[] = []; + // Opens a text segment (start + two deltas), then gets interrupted (no stop / finish received). + for (const e of [ + ev({ event_type: "start", content_items: [] }), + ev({ content_items: [{ type: "text", text: "Par" }] }), + ev({ content_items: [{ type: "text", text: "tial" }] }), + ]) { + for (const m of tr.pushEvent(e)) out.push(m); + } + for (const m of tr.finishInterrupted("timeout")) out.push(m); + + const types = out.map((m) => (m.payload as { type: string }).type); + expect(types).toEqual([ + "partial_text", // start + "partial_text", // delta Par + "partial_text", // delta tial + "partial_text", // stop (backfilled by finishInterrupted) + "text", // complete message + ]); + expect(types).not.toContain("token_usage"); // An interrupted Request has no usage. + + const stop = out[3]!.payload as { event_type: string; stop_reason: string }; + expect(stop.event_type).toBe("stop"); + expect(stop.stop_reason).toBe("timeout"); + const complete = out[4]!.payload as TextPayload; + expect(complete.text).toBe("Partial"); + expect(complete.stop_reason).toBe("timeout"); + }); + + it("closes an open thinking segment with the interruption reason on both partial stop and complete message", () => { + const tr = new EventTranslator(); + const out: OmniMessage[] = []; + // Opens a thinking segment (start + delta), then gets interrupted (no stop / finish received). + for (const e of [ + ev({ event_type: "start", content_items: [] }), + ev({ content_items: [{ type: "thinking", thinking: "half a thought" }] }), + ]) { + for (const m of tr.pushEvent(e)) out.push(m); + } + for (const m of tr.finishInterrupted("aborted")) out.push(m); + + const stop = out.find( + (m) => + (m.payload as { type: string; event_type?: string }).type === "partial_thinking" && + (m.payload as { event_type?: string }).event_type === "stop", + )!.payload as { stop_reason: string }; + expect(stop.stop_reason).toBe("aborted"); + const complete = out.find((m) => (m.payload as { type: string }).type === "thinking")! + .payload as ThinkingPayload; + expect(complete.thinking).toBe("half a thought"); + // Streamed concatenation == complete message: the complete thinking's stop_reason matches + // partial(stop), no longer hardcoded to completed (regression: flushThinking used to + // hardcode completed). + expect(complete.stop_reason).toBe("aborted"); + }); + + it("completes an incomplete (partials-only) tool_call with the interruption reason, not 'completed'", () => { + const tr = new EventTranslator(); + const out: OmniMessage[] = []; + for (const e of [ + ev({ + event_type: "start", + content_items: [ + { type: "partial_tool_call", name: "exec_command", arguments: "", tool_call_id: "c1" }, + ], + }), + ev({ + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"cmd":"ls', + tool_call_id: "c1", + }, + ], + }), + ]) { + for (const m of tr.pushEvent(e)) out.push(m); + } + for (const m of tr.finishInterrupted("aborted")) out.push(m); + + const complete = out.find((m) => (m.payload as { type: string }).type === "tool_call")! + .payload as ToolCallPayload; + expect(complete.tool_call_id).toBe("c1"); + // Key point: not "completed" -> context_engine will not dispatch it for execution (it only + // serves structural completeness and observability). + expect(complete.stop_reason).toBe("aborted"); + expect(complete.arguments).toBe('{"cmd":"ls'); // Keeps the (incomplete) delta accumulated so far. + + const toolStop = out.find( + (m) => + (m.payload as { type: string }).type === "partial_tool_call" && + (m.payload as { event_type: string }).event_type === "stop", + )!.payload as { stop_reason: string }; + expect(toolStop.stop_reason).toBe("aborted"); + expect(out.map((m) => (m.payload as { type: string }).type)).not.toContain("token_usage"); + }); + + it("does not re-emit nor relabel a tool_call already completed mid-stream (keeps 'completed')", () => { + const tr = new EventTranslator(); + const out: OmniMessage[] = []; + for (const e of [ + ev({ + event_type: "start", + content_items: [ + { + type: "partial_tool_call", + name: "exec_command", + arguments: '{"cmd":"ls"}', + tool_call_id: "c1", + }, + ], + }), + ev({ + content_items: [ + { type: "tool_call", name: "exec_command", arguments: { cmd: "ls" }, tool_call_id: "c1" }, + ], + }), + ]) { + for (const m of tr.pushEvent(e)) out.push(m); + } + const before = out.length; + for (const m of tr.finishInterrupted("timeout")) out.push(m); + expect(out.length).toBe(before); // Already produced immediately, not duplicated. + const complete = out.find((m) => (m.payload as { type: string }).type === "tool_call")! + .payload as ToolCallPayload; + expect(complete.stop_reason).toBe("completed"); + }); +}); + +describe("config helpers", () => { + it("maps thinking levels", () => { + expect(mapThinkingLevel("none")).toBe(ThinkingLevel.NONE); + expect(mapThinkingLevel("low")).toBe(ThinkingLevel.LOW); + expect(mapThinkingLevel("medium")).toBe(ThinkingLevel.MEDIUM); + expect(mapThinkingLevel("high")).toBe(ThinkingLevel.HIGH); + expect(mapThinkingLevel("xhigh")).toBe(ThinkingLevel.XHIGH); + expect(mapThinkingLevel(undefined)).toBeUndefined(); + }); + + it("maps tool definitions to schemas (omitting undefined parameters)", () => { + const schemas = toolDefinitionsToSchemas([ + { name: "a", description: "desc a", parameters: { type: "object" } }, + { name: "b", description: "desc b" }, + ]); + expect(schemas[0]).toEqual({ + name: "a", + description: "desc a", + parameters: { type: "object" }, + }); + expect(schemas[1]).toEqual({ name: "b", description: "desc b" }); + expect("parameters" in schemas[1]!).toBe(false); + }); + + it("builds UniConfig with only provided fields", () => { + const cfg = buildUniConfig({ + modelId: "claude-sonnet-4-6", + tools: [{ name: "t", description: "d" }], + systemPrompt: "You are concise.", + maxTokens: 256, + thinkingLevel: "high", + }); + expect(cfg.system_prompt).toBe("You are concise."); + expect(cfg.max_tokens).toBe(256); + expect(cfg.thinking_level).toBe(ThinkingLevel.HIGH); + expect(cfg.tools).toEqual([{ name: "t", description: "d" }]); + + const minimal = buildUniConfig({ modelId: "m", tools: [] }); + expect(minimal.tools).toEqual([]); + expect("system_prompt" in minimal).toBe(false); + expect("max_tokens" in minimal).toBe(false); + expect("thinking_level" in minimal).toBe(false); + }); +}); + +describe("isRetryableError", () => { + it("treats 429, 408 and 5xx as retryable", () => { + expect(isRetryableError({ status: 429 })).toBe(true); + expect(isRetryableError({ status: 408 })).toBe(true); // Request Timeout (transient) + expect(isRetryableError({ status: 500 })).toBe(true); + expect(isRetryableError({ statusCode: 503 })).toBe(true); + }); + + it("treats 4xx auth/param errors as non-retryable", () => { + expect(isRetryableError({ status: 400 })).toBe(false); + expect(isRetryableError({ status: 401 })).toBe(false); + expect(isRetryableError({ status: 403 })).toBe(false); + expect(isRetryableError({ status: 404 })).toBe(false); + }); + + it("treats network error codes as retryable", () => { + expect(isRetryableError({ code: "ECONNRESET" })).toBe(true); + expect(isRetryableError({ code: "ETIMEDOUT" })).toBe(true); + expect(isRetryableError(new Error("socket hang up"))).toBe(true); + expect(isRetryableError(new Error("request timeout"))).toBe(true); + }); + + it("does not retry abort or unknown local errors", () => { + const abort = new Error("aborted"); + abort.name = "AbortError"; + expect(isRetryableError(abort)).toBe(false); + expect(isRetryableError(new Error("unexpected token in JSON"))).toBe(false); + expect(isRetryableError(null)).toBe(false); + expect(isRetryableError(undefined)).toBe(false); + }); +}); + +describe("isMalformedJsonParseError", () => { + it("detects JSON.parse SyntaxError by exception type, including the cause chain", () => { + // AgentHub uses JSON.parse internally; a parse failure throws a SyntaxError, so it can be + // determined directly by exception type. + expect( + isMalformedJsonParseError(new SyntaxError("Unexpected token < in JSON at position 0")), + ).toBe(true); + // An error wrapped by a higher layer can still be determined via the cause chain. + expect( + isMalformedJsonParseError( + new Error("request failed", { + cause: new SyntaxError("Unexpected end of JSON input"), + }), + ), + ).toBe(true); + // A non-SyntaxError does not count as malformed (even if the message mentions JSON), + // leaving classification to the network/failure path. + expect(isMalformedJsonParseError(new Error("Unexpected token < in JSON at position 0"))).toBe( + false, + ); + expect(isMalformedJsonParseError(new Error("socket hang up"))).toBe(false); + }); +}); + +describe("isIncompleteStreamError", () => { + it("detects AgentHub incomplete-stream validation errors by message prefix, incl. cause chain", () => { + // The server/proxy cleanly terminates the stream early at an event boundary: AgentHub's + // final-event validation throws a plain Error. + expect(isIncompleteStreamError(new Error("Streaming response yielded no events"))).toBe(true); + expect( + isIncompleteStreamError(new Error('Last event must carry usage_metadata, got: {"a":1}')), + ).toBe(true); + expect(isIncompleteStreamError(new Error("Last event must carry finish_reason, got: {}"))).toBe( + true, + ); + expect( + isIncompleteStreamError( + new Error("request failed", { cause: new Error("Streaming response yielded no events") }), + ), + ).toBe(true); + expect(isIncompleteStreamError(new Error("socket hang up"))).toBe(false); + expect(isIncompleteStreamError(null)).toBe(false); + }); +}); + +describe("GenerativeModel.streamGenerate outcome classification (PRN-013)", () => { + // Injects a controlled UniEvent stream through the protected openStream seam to verify the + // outcome classification of timeout/network-drop/interrupt/error, without needing a real API. + // Construction only creates the config object; no network involved. + class SeamModel extends GenerativeModel { + constructor( + private readonly source: (signal: AbortSignal) => AsyncIterable, + timeoutMs = 10000, + ) { + super({ modelId: "claude-sonnet-4-6", tools: [], requestTimeoutMs: timeoutMs }); + } + protected override openStream(_uni: UniMessage, signal: AbortSignal): AsyncIterable { + return this.source(signal); + } + } + + const abortError = (): Error => Object.assign(new Error("aborted"), { name: "AbortError" }); + + // Never yields any event, and only ends with an AbortError once the signal aborts + // (simulates idle/hanging). + async function* hang(signal: AbortSignal): AsyncGenerator { + await new Promise((_, reject) => { + if (signal.aborted) { + reject(abortError()); + return; + } + signal.addEventListener("abort", () => reject(abortError()), { once: true }); + }); + } + + // Yields one piece of text, then throws a retryable network error (network drop). + async function* dropAfterText(): AsyncGenerator { + yield ev({ content_items: [{ type: "text", text: "hi" }] }); + throw Object.assign(new Error("socket hang up"), { code: "ECONNRESET" }); + } + + // Immediately throws a non-retryable error (auth). + async function* authError(): AsyncGenerator { + throw Object.assign(new Error("invalid api key"), { status: 401 }); + } + + // AgentHub's response body is not valid JSON (e.g. the gateway returns HTML / a truncated response). + async function* malformedJsonAfterText(): AsyncGenerator { + yield ev({ content_items: [{ type: "text", text: "hi" }] }); + throw new SyntaxError("Unexpected token < in JSON at position 0"); + } + + const typeOf = (m: OmniMessage): string => (m.payload as { type?: string }).type ?? ""; + + async function drain( + gen: AsyncGenerator, + ): Promise<{ messages: OmniMessage[]; outcome: LLMOutcome }> { + const messages: OmniMessage[] = []; + let res = await gen.next(); + while (!res.done) { + messages.push(res.value); + res = await gen.next(); + } + return { messages, outcome: res.value as LLMOutcome }; + } + + it("returns failed (never throws) on a build failure such as empty input", async () => { + const model = new SeamModel((sig) => hang(sig)); + const { messages, outcome } = await drain(model.streamGenerate({ newMessages: [] })); + expect(outcome.status).toBe("failed"); // A mergeOmniToUniMessage failure converges to failed, never throws + expect(messages).toHaveLength(0); + }); + + it("中断落在消费者挂起于 yield 期间:立即以 aborted 收尾,绝不再去拉已被 abort 的上游", async () => { + // This is exactly the cause of "the session hangs forever after interrupting it in the + // browser": when the user interrupts, this generator is usually suspended at `yield` + // (the engine is blocked on `await approve(tc)` waiting for manual approval). onUserAbort + // has already aborted the upstream stream; when the consumer comes to pull again, if we go + // back and call `it.next()` on that now-dead stream, the promise will never settle again -- + // and the idle timer cannot save it either (once it fires, it just aborts again, which is a + // no-op on an already-aborted stream). The run then never finishes, and the Session is stuck + // in running: it can neither send messages nor compact (the frontend's /compact is gated by + // !running, so clicking it does nothing). + // + // The upstream simulates the real cancellation behavior with "pulling again after being + // aborted never settles." If the fix is missing, this test hangs until it times out and fails. + async function* deadAfterAbort(): AsyncGenerator { + yield ev({ content_items: [{ type: "text", text: "hi" }] }); + await new Promise(() => {}); // Never settles + } + const ac = new AbortController(); + // Give the idle timeout plenty of headroom, so it's the "pre-interrupt check" doing the + // finishing, not the timer as a fallback. + const model = new SeamModel(() => deadAfterAbort(), 60_000); + const gen = model.streamGenerate({ newMessages: [userText("go")], signal: ac.signal }); + + const first = await gen.next(); // Gets the first message -> this generator is now suspended at yield + expect(first.done).toBe(false); + + ac.abort(); // User interrupt (we are suspended at yield right now, not inside it.next()) + + // The already-resolved buffered messages are drained as usual, and afterward it **must** + // finish -- the key point is that it ends, rather than going back to pull that dead + // upstream and hanging the whole run forever (if the fix is missing, this would never + // get a result, and the test times out and fails). + let res = await gen.next(); + while (!res.done) res = await gen.next(); + expect(res.value).toMatchObject({ status: "aborted" }); + }); + + it("classifies an idle timeout as timeout, with no token_usage", async () => { + const model = new SeamModel((sig) => hang(sig), 30); // 30ms idle timeout + const { messages, outcome } = await drain( + model.streamGenerate({ newMessages: [userText("go")] }), + ); + expect(outcome.status).toBe("timeout"); + expect(messages.map(typeOf)).not.toContain("token_usage"); + }); + + it("classifies an idle timeout as timeout even when the stream ends gracefully on abort", async () => { + // The underlying implementation does not throw on abort, and ends gracefully with done: + // this must still be classified as timeout, not mistakenly as completed. + async function* gracefulHang(signal: AbortSignal): AsyncGenerator { + await new Promise((resolve) => { + if (signal.aborted) { + resolve(); + return; + } + signal.addEventListener("abort", () => resolve(), { once: true }); + }); + } + const model = new SeamModel((sig) => gracefulHang(sig), 30); + const { messages, outcome } = await drain( + model.streamGenerate({ newMessages: [userText("go")] }), + ); + expect(outcome.status).toBe("timeout"); + expect(messages.map(typeOf)).not.toContain("token_usage"); + }); + + it("classifies a network drop as timeout, closing the open text segment, no token_usage", async () => { + const model = new SeamModel(() => dropAfterText()); + const { messages, outcome } = await drain( + model.streamGenerate({ newMessages: [userText("go")] }), + ); + expect(outcome.status).toBe("timeout"); + const complete = messages.find((m) => typeOf(m) === "text"); + expect((complete!.payload as TextPayload).text).toBe("hi"); + expect((complete!.payload as TextPayload).stop_reason).toBe("timeout"); + expect(messages.map(typeOf)).not.toContain("token_usage"); + }); + + it("classifies an AgentHub JSON parse exception as malformed and closes partial output", async () => { + const model = new SeamModel(() => malformedJsonAfterText()); + const { messages, outcome } = await drain( + model.streamGenerate({ newMessages: [userText("go")] }), + ); + expect(outcome.status).toBe("malformed"); + expect(outcome.message).toContain("Unexpected token"); + const complete = messages.find((m) => typeOf(m) === "text"); + expect((complete!.payload as TextPayload).text).toBe("hi"); + expect((complete!.payload as TextPayload).stop_reason).toBe("malformed"); + expect(messages.map(typeOf)).not.toContain("token_usage"); + }); + + it("classifies a cleanly-truncated stream (AgentHub last-event validation) as malformed, not failed", async () => { + // The server/proxy cleanly drops the stream at an event boundary (no network error thrown): + // AgentHub's final-event validation throws a plain Error; this is an incomplete LLM + // Request that must go through the malformed reconnect path, and must not abort the task as failed. + async function* cleanTruncationAfterText(): AsyncGenerator { + yield ev({ content_items: [{ type: "text", text: "hi" }] }); + throw new Error('Last event must carry usage_metadata, got: {"content_items":[]}'); + } + const model = new SeamModel(() => cleanTruncationAfterText()); + const { messages, outcome } = await drain( + model.streamGenerate({ newMessages: [userText("go")] }), + ); + expect(outcome.status).toBe("malformed"); + expect(messages.map(typeOf)).not.toContain("token_usage"); + + async function* noEvents(): AsyncGenerator { + throw new Error("Streaming response yielded no events"); + } + const model2 = new SeamModel(() => noEvents()); + const { outcome: outcome2 } = await drain( + model2.streamGenerate({ newMessages: [userText("go")] }), + ); + expect(outcome2.status).toBe("malformed"); + }); + + it("classifies a non-retryable error as failed with message, no token_usage", async () => { + const model = new SeamModel(() => authError()); + const { messages, outcome } = await drain( + model.streamGenerate({ newMessages: [userText("go")] }), + ); + expect(outcome.status).toBe("failed"); + expect(outcome.message).toContain("invalid api key"); + expect(messages.map(typeOf)).not.toContain("token_usage"); + }); + + it("classifies a user abort (mid idle) as aborted, not timeout", async () => { + const controller = new AbortController(); + const model = new SeamModel((sig) => hang(sig)); // Default 10s timeout, won't fire first + const p = drain( + model.streamGenerate({ + newMessages: [userText("go")], + signal: controller.signal, + }), + ); + setTimeout(() => controller.abort(), 20); + const { outcome } = await p; + expect(outcome.status).toBe("aborted"); + }); +}); + +describe("provider fidelity fields (signature / phase)", () => { + const complete = (messages: ReturnType["messages"]) => + messages.filter((m) => !(m.payload as { type: string }).type.startsWith("partial_")); + + it("captures the thinking signature arriving as an empty-text delta (Claude signature_delta)", () => { + const { messages } = translateEvents([ + ev({ content_items: [{ type: "thinking", thinking: "let me think" }] }), + ev({ content_items: [{ type: "thinking", thinking: "", signature: "sig-abc" }] }), + ev({ content_items: [{ type: "text", text: "answer" }] }), + ev({ event_type: "stop", content_items: [], finish_reason: "stop" }), + ]); + const thinking = complete(messages).find( + (m) => (m.payload as { type: string }).type === "thinking", + )!; + expect((thinking.payload as { thinking: string; signature?: string }).thinking).toBe( + "let me think", + ); + expect((thinking.payload as { signature?: string }).signature).toBe("sig-abc"); + }); + + it("splits adjacent thinking blocks on signature (redacted + normal keep their own signatures)", () => { + const { messages } = translateEvents([ + // A redacted block: sentinel text + signature arrive together (Claude content_block_start). + ev({ + content_items: [{ type: "thinking", thinking: "_REDACTED_THINKING", signature: "sig-red" }], + }), + // The next, ordinary thinking block. + ev({ content_items: [{ type: "thinking", thinking: "visible" }] }), + ev({ content_items: [{ type: "thinking", thinking: "", signature: "sig-vis" }] }), + ev({ event_type: "stop", content_items: [], finish_reason: "stop" }), + ]); + const thinkings = complete(messages).filter( + (m) => (m.payload as { type: string }).type === "thinking", + ); + expect( + thinkings.map((m) => { + const p = m.payload as { thinking: string; signature?: string }; + return [p.thinking, p.signature]; + }), + ).toEqual([ + ["_REDACTED_THINKING", "sig-red"], + ["visible", "sig-vis"], + ]); + }); + + it("emits an empty-text thinking with signature (GPT-5 encrypted reasoning)", () => { + const { messages } = translateEvents([ + ev({ content_items: [{ type: "thinking", thinking: "", signature: '{"id":"rs_1"}' }] }), + ev({ content_items: [{ type: "text", text: "answer" }] }), + ev({ event_type: "stop", content_items: [], finish_reason: "stop" }), + ]); + const thinking = complete(messages).find( + (m) => (m.payload as { type: string }).type === "thinking", + )!; + expect((thinking.payload as { thinking: string }).thinking).toBe(""); + expect((thinking.payload as { signature?: string }).signature).toBe('{"id":"rs_1"}'); + }); + + it("splits text segments on phase markers arriving as empty-text deltas (GPT-5)", () => { + const { messages } = translateEvents([ + ev({ content_items: [{ type: "text", text: "", phase: "planning" }] }), + ev({ content_items: [{ type: "text", text: "plan..." }] }), + ev({ content_items: [{ type: "text", text: "", phase: "answer" }] }), + ev({ content_items: [{ type: "text", text: "final" }] }), + ev({ event_type: "stop", content_items: [], finish_reason: "stop" }), + ]); + const texts = complete(messages).filter((m) => (m.payload as { type: string }).type === "text"); + expect( + texts.map((m) => { + const p = m.payload as { text: string; phase?: string | null }; + return [p.text, p.phase]; + }), + ).toEqual([ + ["plan...", "planning"], + ["final", "answer"], + ]); + }); + + it("carries the tool_call signature through to the complete message", () => { + const { messages } = translateEvents([ + ev({ + content_items: [ + { + type: "tool_call", + name: "exec_command", + arguments: { cmd: "ls" }, + tool_call_id: "tc1", + signature: "sig-tool", + }, + ], + }), + ev({ event_type: "stop", content_items: [], finish_reason: "tool_call" }), + ]); + const tc = complete(messages).find( + (m) => (m.payload as { type: string }).type === "tool_call", + )!; + expect((tc.payload as { signature?: string }).signature).toBe("sig-tool"); + }); + + it("round-trips fidelity fields back to UniMessage content items (setHistory path)", () => { + const uni = mergeOmniToUniMessage([ + thinkingMessage("deep", "completed", { signature: "sig-1" }), + assistantText("hi", "completed", { phase: "answer", signature: "sig-2" }), + toolCall({ name: "t", arguments: "{}", toolCallId: "tc1", signature: "sig-3" }), + ]); + expect(uni.content_items).toEqual([ + { type: "thinking", thinking: "deep", signature: "sig-1" }, + { type: "text", text: "hi", phase: "answer", signature: "sig-2" }, + { type: "tool_call", name: "t", arguments: {}, tool_call_id: "tc1", signature: "sig-3" }, + ]); + }); +}); + +describe("flushText signature parity (PR #39 review)", () => { + it("emits an empty-text message carrying a text signature instead of dropping it", () => { + const { messages } = translateEvents([ + ev({ content_items: [{ type: "text", text: "", signature: "sig-t" }] }), + ev({ event_type: "stop", content_items: [], finish_reason: "stop" }), + ]); + const text = messages.find((m) => (m.payload as { type: string }).type === "text")!; + expect((text.payload as { text: string }).text).toBe(""); + expect((text.payload as { signature?: string }).signature).toBe("sig-t"); + }); +}); + +describe("tool_call_id 唯一化(name-as-id provider,如 Gemini 以函数名当 id)", () => { + const callIdsOf = (messages: OmniMessage[]): string[] => + messages + .filter((m) => (m.payload as { type: string }).type === "tool_call") + .map((m) => (m.payload as ToolCallPayload).tool_call_id); + + it("同一 Request 内重复 id 的第二个完整 tool_call 不被丢弃,且分到 #2 后缀", () => { + const { messages } = translateEvents([ + ev({ + event_type: "stop", + finish_reason: "tool_call", + content_items: [ + { + type: "tool_call", + name: "get_time", + arguments: { city: "Tokyo" }, + tool_call_id: "get_time", + }, + { + type: "tool_call", + name: "get_time", + arguments: { city: "Paris" }, + tool_call_id: "get_time", + }, + ], + }), + ]); + expect(callIdsOf(messages)).toEqual(["get_time", "get_time#2"]); + const calls = messages + .filter((m) => (m.payload as { type: string }).type === "tool_call") + .map((m) => m.payload as ToolCallPayload); + expect(calls[0]!.arguments).toBe('{"city":"Tokyo"}'); + expect(calls[1]!.arguments).toBe('{"city":"Paris"}'); + expect(calls.every((c) => c.stop_reason === "completed")).toBe(true); + }); + + it("不同 id 的并行调用不受影响(原样透传,无后缀)", () => { + const { messages } = translateEvents([ + ev({ + event_type: "stop", + finish_reason: "tool_call", + content_items: [ + { type: "tool_call", name: "get_time", arguments: {}, tool_call_id: "get_time" }, + { type: "tool_call", name: "get_weather", arguments: {}, tool_call_id: "get_weather" }, + { type: "tool_call", name: "get_time", arguments: {}, tool_call_id: "get_time" }, + ], + }), + ]); + expect(callIdsOf(messages)).toEqual(["get_time", "get_weather", "get_time#2"]); + }); + + it("跨 Request 共享登记表:下一轮同名调用分到新后缀(前端工具卡不再互相覆盖)", () => { + const ids = new ToolCallIdAllocator(); + const round = (city: string): string[] => { + const translator = new EventTranslator(ids); + const out: OmniMessage[] = []; + const event = ev({ + event_type: "stop", + finish_reason: "tool_call", + content_items: [ + { type: "tool_call", name: "get_time", arguments: { city }, tool_call_id: "get_time" }, + ], + }); + for (const m of translator.pushEvent(event)) out.push(m); + for (const m of translator.finish()) out.push(m); + return callIdsOf(out); + }; + expect(round("Tokyo")).toEqual(["get_time"]); + expect(round("Paris")).toEqual(["get_time#2"]); + expect(round("NYC")).toEqual(["get_time#3"]); + }); + + it("跨 Request 撞车时,partial 片段与完整消息用同一个带后缀 id", () => { + const ids = new ToolCallIdAllocator(); + ids.markUsed("exec"); // this provider id was already taken in the previous turn + const translator = new EventTranslator(ids); + const out: OmniMessage[] = []; + const push = (e: UniEvent): void => { + for (const m of translator.pushEvent(e)) out.push(m); + }; + push( + ev({ + event_type: "start", + content_items: [ + { type: "partial_tool_call", name: "exec", arguments: '{"cmd":', tool_call_id: "exec" }, + ], + }), + ); + push( + ev({ + content_items: [ + { type: "partial_tool_call", name: "", arguments: '"ls"}', tool_call_id: "exec" }, + ], + }), + ); + push( + ev({ + event_type: "stop", + finish_reason: "tool_call", + content_items: [ + { type: "tool_call", name: "exec", arguments: { cmd: "ls" }, tool_call_id: "exec" }, + ], + }), + ); + for (const m of translator.finish()) out.push(m); + + const partialIds = out + .filter((m) => (m.payload as { type: string }).type === "partial_tool_call") + .map((m) => (m.payload as { tool_call_id: string }).tool_call_id); + expect(partialIds.length).toBeGreaterThanOrEqual(3); // start + delta×2 + stop + expect(partialIds.every((id) => id === "exec#2")).toBe(true); + expect(callIdsOf(out)).toEqual(["exec#2"]); + }); + + it("出站还原:tool_call / tool_call_output 的 #n 后缀发往 provider 前剥掉,无后缀原样", () => { + const result = mergeOmniToUniMessage([ + toolCallOutput({ output: "10:00", toolCallId: "get_time#2" }), + ]); + expect((result.content_items[0] as { tool_call_id: string }).tool_call_id).toBe("get_time"); + + const call = mergeOmniToUniMessage([ + toolCall({ name: "get_time", arguments: "{}", toolCallId: "get_time#3" }), + ]); + expect((call.content_items[0] as { tool_call_id: string }).tool_call_id).toBe("get_time"); + + const passthrough = mergeOmniToUniMessage([ + toolCallOutput({ output: "ok", toolCallId: "call_Ab12" }), + ]); + expect((passthrough.content_items[0] as { tool_call_id: string }).tool_call_id).toBe( + "call_Ab12", + ); + }); + + it("stripToolCallIdSuffix 只剥结尾的 #数字(幂等,不误伤中缀)", () => { + expect(stripToolCallIdSuffix("get_time#2")).toBe("get_time"); + expect(stripToolCallIdSuffix("get_time#12")).toBe("get_time"); + expect(stripToolCallIdSuffix("get_time")).toBe("get_time"); + expect(stripToolCallIdSuffix("a#2b")).toBe("a#2b"); + expect(stripToolCallIdSuffix("a#x")).toBe("a#x"); + }); + + it("ToolCallIdAllocator:探测跳过已占用后缀;markUsed 播种生效", () => { + const ids = new ToolCallIdAllocator(); + ids.markUsed("t"); + ids.markUsed("t#2"); + expect(ids.allocate("t")).toBe("t#3"); + expect(ids.allocate("u")).toBe("u"); + expect(ids.allocate("u")).toBe("u#2"); + }); + + it("恢复播种:setHistory 后新的同名调用不与历史撞 id", async () => { + class SeedModel extends GenerativeModel { + constructor() { + super({ modelId: "claude-sonnet-4-6", tools: [] }); + } + protected override openStream( + _uni: UniMessage, + _signal: AbortSignal, + ): AsyncIterable { + return (async function* () { + yield ev({ + event_type: "stop", + finish_reason: "tool_call", + content_items: [ + { + type: "tool_call", + name: "get_time", + arguments: { city: "Paris" }, + tool_call_id: "get_time", + }, + ], + }); + })(); + } + } + const model = new SeedModel(); + model.setHistory([ + userText("What time is it in Tokyo?"), + toolCall({ name: "get_time", arguments: '{"city":"Tokyo"}', toolCallId: "get_time" }), + toolCallOutput({ output: "10:00", toolCallId: "get_time" }), + ]); + + const out: OmniMessage[] = []; + const gen = model.streamGenerate({ newMessages: [userText("And Paris?")] }); + let res = await gen.next(); + while (!res.done) { + out.push(res.value); + res = await gen.next(); + } + expect(callIdsOf(out)).toEqual(["get_time#2"]); + }); +}); diff --git a/packages/core/test/model-catalog.test.ts b/packages/core/test/model-catalog.test.ts new file mode 100644 index 0000000..a618dd8 --- /dev/null +++ b/packages/core/test/model-catalog.test.ts @@ -0,0 +1,212 @@ +/** + * Built-in model catalog unit tests: unique ids, valid provider references, positive + * three-bucket pricing, lookups, and preset entry generation. + */ +import { describe, expect, it } from "vitest"; +import { + MODEL_CATALOG, + MODEL_PROVIDERS, + catalogEntryFor, + inferProviderForUpstream, + presetModelEntries, + providerInfo, + resolveModelEnv, +} from "../src/state/index.js"; + +describe("model-catalog", () => { + it("model id 全局唯一;DeepSeek 排在最前(默认模型所属厂商)", () => { + const ids = MODEL_CATALOG.map((m) => m.modelId); + expect(new Set(ids).size).toBe(ids.length); + expect(MODEL_CATALOG[0]!.provider).toBe("deepseek"); + // Group order: DeepSeek first, followed by the OpenRouter and SiliconFlow gateways, + // then Google Gemini before Anthropic, with custom last. + expect(MODEL_PROVIDERS.map((p) => p.id)).toEqual([ + "deepseek", + "openrouter", + "siliconflow", + "google", + "anthropic", + "openai", + "zhipu", + "moonshot", + "custom", + ]); + expect(providerInfo("siliconflow")!.label).toBe("SiliconFlow"); + // The catalog no longer includes GLM-5-Turbo. + expect(ids).not.toContain("glm-5-turbo"); + }); + + it("provider 均在 MODEL_PROVIDERS 内(custom 仅收拢自定义模型,目录不使用)", () => { + const providerIds = new Set(MODEL_PROVIDERS.map((p) => p.id)); + for (const m of MODEL_CATALOG) { + expect(providerIds.has(m.provider)).toBe(true); + expect(m.provider).not.toBe("custom"); + } + // Except for custom, every provider gives a console link to "get an API key" (shown in the + // frontend's group header) and a model list / docs link to "get a model id" (shown in the add-model dialog). + for (const p of MODEL_PROVIDERS) { + if (p.id === "custom") { + expect(p.apiKeyUrl).toBeUndefined(); + expect(p.modelsUrl).toBeUndefined(); + } else { + expect(p.apiKeyUrl).toMatch(/^https:\/\//); + expect(p.modelsUrl).toMatch(/^https:\/\//); + } + } + // Provider ids are also unique, and each provider has an API key / base URL env var name. + expect(new Set([...providerIds]).size).toBe(MODEL_PROVIDERS.length); + for (const p of MODEL_PROVIDERS) { + expect(p.envKey).toMatch(/_API_KEY$/); + expect(p.envBaseUrlKey).toMatch(/_BASE_URL$/); + } + }); + + it("三桶价格均为正数,context_window 为正整数", () => { + for (const m of MODEL_CATALOG) { + expect(m.pricing, m.modelId).toBeDefined(); + expect(m.pricing!.unit).toBe("usd_per_mtok"); + expect(m.pricing!.cache_read).toBeGreaterThan(0); + expect(m.pricing!.cache_write).toBeGreaterThan(0); + expect(m.pricing!.output).toBeGreaterThan(0); + expect(Number.isInteger(m.contextWindow)).toBe(true); + expect(m.contextWindow!).toBeGreaterThan(0); + } + }); + + it("providerInfo 按 id 命中,未知返回 undefined", () => { + expect(providerInfo("moonshot")?.envKey).toBe("MOONSHOT_API_KEY"); + expect(providerInfo("nonexistent")).toBeUndefined(); + }); + + it("成对匹配与分组推断:catalogEntryFor / inferProviderForUpstream", () => { + // catalogEntryFor is the sole catalog lookup entry point: it matches on (group, upstream id) + // pairs, so an identically named upstream id never matches across the wrong group. + expect(catalogEntryFor("anthropic", "claude-sonnet-4-6")?.displayName).toBe( + "Claude Sonnet 4.6", + ); + expect(catalogEntryFor("openai", "claude-sonnet-4-6")).toBeUndefined(); + // The upstream id itself may contain / (gateway models); it is never split apart. + expect(catalogEntryFor("openrouter", "xiaomi/mimo-v2.5")?.displayName).toBe("MiMo-V2.5"); + expect(catalogEntryFor("custom", "my-own")).toBeUndefined(); + + // Inference for `model add` when --provider is omitted: a catalog hit yields its provider, otherwise custom. + expect(inferProviderForUpstream("deepseek-v4-pro")).toBe("deepseek"); + expect(inferProviderForUpstream("xiaomi/mimo-v2.5")).toBe("openrouter"); + expect(inferProviderForUpstream("my-own-model")).toBe("custom"); + }); + + it("presetModelEntries:provider 与纯上游 model_id 分列;网关模型内联 base_url", () => { + const entries = presetModelEntries(); + expect(entries).toHaveLength(MODEL_CATALOG.length); + for (const [i, entry] of entries.entries()) { + const cat = MODEL_CATALOG[i]!; + expect(entry.provider).toBe(cat.provider); + expect(entry.model_id).toBe(cat.modelId); + expect(entry.context_window).toBe(cat.contextWindow); + expect(entry.pricing).toEqual(cat.pricing); + expect(entry.vision).toBe(cat.supportsVision ? undefined : false); + // Models that AgentHub can auto-route leave client_type unset; OpenRouter gateway models set it to openai. + expect(entry.client_type).toBe(cat.clientType); + // Gateway models inline a preset base URL (no credentials); other models carry no credential at all. + expect(entry.base_url).toBe(cat.baseUrl); + expect(entry.api_key).toBeUndefined(); + // The concatenated storage id and request_model_id have been removed and no longer appear. + expect(Object.hasOwn(entry, "request_model_id")).toBe(false); + } + }); + + it("网关模型(OpenRouter / SiliconFlow):openai 协议 + 预置 base URL;env 兜底为 OPENAI_API_KEY", () => { + const or = MODEL_CATALOG.filter((m) => m.provider === "openrouter"); + expect(or.map((m) => m.modelId)).toEqual([ + "xiaomi/mimo-v2.5", + "tencent/hy3", + "minimax/minimax-m3", + "stepfun/step-3.7-flash", + ]); + for (const m of or) { + expect(m.clientType).toBe("openai"); + expect(m.baseUrl).toBe("https://openrouter.ai/api/v1"); + } + const sf = MODEL_CATALOG.filter((m) => m.provider === "siliconflow"); + expect(sf.map((m) => m.modelId)).toEqual([ + "zai-org/GLM-5.2", + "deepseek-ai/DeepSeek-V4-Pro", + "meituan-longcat/LongCat-2.0", + ]); + for (const m of sf) { + expect(m.clientType).toBe("openai"); + expect(m.baseUrl).toBe("https://api.siliconflow.cn/v1"); + } + // Routed through AgentHub's OpenAI client -> when the credential is left blank it reads OPENAI_API_KEY (not the provider's own env var name). + for (const id of ["openrouter", "siliconflow", "custom"]) { + expect(providerInfo(id)!.envKey).toBe("OPENAI_API_KEY"); + expect(providerInfo(id)!.envBaseUrlKey).toBe("OPENAI_BASE_URL"); + } + // gatewayBaseUrl (prefilled by group in the frontend's "add model" dialog) is only carried by the two gateway providers. + expect(providerInfo("openrouter")!.gatewayBaseUrl).toBe("https://openrouter.ai/api/v1"); + expect(providerInfo("siliconflow")!.gatewayBaseUrl).toBe("https://api.siliconflow.cn/v1"); + for (const p of MODEL_PROVIDERS) { + if (p.id !== "openrouter" && p.id !== "siliconflow") { + expect(p.gatewayBaseUrl, p.id).toBeUndefined(); + } + } + const gateway = [...or, ...sf]; + // Pricing (USD): MiMo v2.5 and Hy3. + const mimo = MODEL_CATALOG.find((m) => m.modelId === "xiaomi/mimo-v2.5")!.pricing!; + expect([mimo.cache_read, mimo.cache_write, mimo.output]).toEqual([0.0028, 0.14, 0.28]); + const hy3 = MODEL_CATALOG.find((m) => m.modelId === "tencent/hy3")!.pricing!; + expect([hy3.cache_read, hy3.cache_write, hy3.output]).toEqual([0.035, 0.14, 0.58]); + + // In preset entries, exactly the gateway models (and only them) inline base_url (no credentials). + const withBaseUrl = presetModelEntries().filter((e) => e.base_url !== undefined); + expect(withBaseUrl.map((e) => [e.provider, e.model_id]).sort()).toEqual( + gateway.map((m) => [m.provider, m.modelId]).sort(), + ); + }); + + it("DeepSeek 与 Kimi 按官方人民币价初始化(存储美元,×7 还原官方原价)", () => { + const cnyOf = (usdV: number) => Math.round(usdV * 7 * 1000) / 1000; + const pro = MODEL_CATALOG.find((m) => m.modelId === "deepseek-v4-pro")!.pricing!; + expect([cnyOf(pro.cache_read), cnyOf(pro.cache_write), cnyOf(pro.output)]).toEqual([ + 0.025, 3, 6, + ]); + const k26 = MODEL_CATALOG.find((m) => m.modelId === "kimi-k2.6")!.pricing!; + expect([cnyOf(k26.cache_read), cnyOf(k26.cache_write), cnyOf(k26.output)]).toEqual([ + 1.1, 6.5, 27, + ]); + }); +}); + +describe("resolveModelEnv(PRN-021:env 兜底按 AgentHub 路由规则解析)", () => { + it("一方厂商 id 按子串路由到厂商客户端的环境变量", () => { + expect(resolveModelEnv("deepseek-v4-pro")?.envKey).toBe("DEEPSEEK_API_KEY"); + expect(resolveModelEnv("claude-opus-4-8")?.envKey).toBe("ANTHROPIC_API_KEY"); + expect(resolveModelEnv("claude-sonnet-4-6")?.envKey).toBe("ANTHROPIC_API_KEY"); + expect(resolveModelEnv("gemini-3.5-flash")?.envKey).toBe("GEMINI_API_KEY"); + expect(resolveModelEnv("gpt-5.5-pro")?.envKey).toBe("OPENAI_API_KEY"); + expect(resolveModelEnv("glm-5.2")?.envKey).toBe("ZAI_API_KEY"); + expect(resolveModelEnv("kimi-k2.6")?.envBaseUrlKey).toBe("MOONSHOT_BASE_URL"); + }); + + it("显式 client_type 优先于 id:openai 协议一律 OPENAI_*(与分组无关)", () => { + expect(resolveModelEnv("deepseek-v4-pro", "openai")?.envKey).toBe("OPENAI_API_KEY"); + expect(resolveModelEnv("zai-org/GLM-5.2", "openai")?.envKey).toBe("OPENAI_API_KEY"); + }); + + it("无法路由的 id 返回 undefined(AgentHub 会拒绝,需显式 client_type 或 OpenAI 协议分组)", () => { + expect(resolveModelEnv("totally-unknown-model")).toBeUndefined(); + expect(resolveModelEnv("xiaomi/mimo-v2.5")).toBeUndefined(); + }); + + it("目录不变式:无 client_type 的条目按 id 可路由且 envKey 与厂商一致;网关条目经 client_type 解析为 OPENAI_*", () => { + for (const m of MODEL_CATALOG) { + const env = resolveModelEnv(m.modelId, m.clientType); + expect(env, `${m.provider}/${m.modelId}`).toBeDefined(); + if (m.clientType === undefined) { + expect(env!.envKey, m.modelId).toBe(providerInfo(m.provider)!.envKey); + } else { + expect(env!.envKey, m.modelId).toBe("OPENAI_API_KEY"); + } + } + }); +}); diff --git a/packages/core/test/omnimessage.test.ts b/packages/core/test/omnimessage.test.ts new file mode 100644 index 0000000..2152ff1 --- /dev/null +++ b/packages/core/test/omnimessage.test.ts @@ -0,0 +1,198 @@ +import { describe, expect, it } from "vitest"; +import { + PartialAggregator, + aggregateAll, + abortEvent, + addTokenCounts, + approvalDecision, + assistantText, + emptyTokenCounts, + isCompleteModelMessage, + partialText, + partialToolCall, + partialToolCallOutput, + toolCall, + toolCallOutput, + userText, +} from "../src/omnimessage/index.js"; +import type { + TextPayload, + ToolCallOutputPayload, + ToolCallPayload, +} from "../src/omnimessage/index.js"; + +describe("builders", () => { + it("stamps an ISO 8601 timestamp and correct shells", () => { + const m = userText("hi"); + expect(m.type).toBe("model_msg"); + expect(m.payload.type).toBe("text"); + expect(m.payload.role).toBe("user"); + expect(new Date(m.timestamp).toISOString()).toBe(m.timestamp); + }); + + it("builds event messages", () => { + const a = approvalDecision("allow", "call_1"); + expect(a.type).toBe("event_msg"); + expect(a.payload).toMatchObject({ + type: "approval_decision", + decision: "allow", + tool_call_id: "call_1", + }); + expect(abortEvent("stop").payload).toMatchObject({ + type: "abort", + reason: "stop", + }); + }); + + it("toolCallOutput carries optional images and round-trips through JSON", () => { + const dataUrl = "data:image/png;base64,AAAA"; + const msg = toolCallOutput({ + output: "image/png, 4 B", + toolCallId: "call_img", + images: [dataUrl], + }); + expect(msg.payload).toMatchObject({ + type: "tool_call_output", + role: "user", + output: "image/png, 4 B", + images: [dataUrl], + tool_call_id: "call_img", + stop_reason: "completed", + }); + // A JSON serialization round-trip preserves images (same shape as Trace persistence / replay). + const revived = JSON.parse(JSON.stringify(msg)) as { payload: ToolCallOutputPayload }; + expect(revived.payload.images).toEqual([dataUrl]); + // The images field is not produced when omitted or given an empty array (absence means no + // images; serialization does not carry an empty field). + expect("images" in toolCallOutput({ output: "x", toolCallId: "c" }).payload).toBe(false); + expect("images" in toolCallOutput({ output: "x", toolCallId: "c", images: [] }).payload).toBe( + false, + ); + }); + + it("partialToolCallOutput delta carries optional images (whole image in one delta)", () => { + const dataUrl = "data:image/png;base64,AAAA"; + const msg = partialToolCallOutput({ + eventType: "delta", + toolCallId: "call_img", + images: [dataUrl], + }); + expect(msg.payload).toMatchObject({ + type: "partial_tool_call_output", + event_type: "delta", + output: "", + images: [dataUrl], + tool_call_id: "call_img", + }); + // The images field is not produced when omitted or given an empty array (same as the complete message). + expect("images" in partialToolCallOutput({ eventType: "delta", toolCallId: "c" }).payload).toBe( + false, + ); + expect( + "images" in + partialToolCallOutput({ eventType: "delta", toolCallId: "c", images: [] }).payload, + ).toBe(false); + }); + + it("adds token counts", () => { + const a = { cache_read: 1, cache_write: 2, output: 3, total: 6 }; + const b = { cache_read: 10, cache_write: 20, output: 30, total: 60 }; + expect(addTokenCounts(a, b)).toEqual({ + cache_read: 11, + cache_write: 22, + output: 33, + total: 66, + }); + expect(emptyTokenCounts()).toEqual({ + cache_read: 0, + cache_write: 0, + output: 0, + total: 0, + }); + }); +}); + +describe("isCompleteModelMessage", () => { + it("distinguishes complete from partial model messages", () => { + expect(isCompleteModelMessage(assistantText("done"))).toBe(true); + expect(isCompleteModelMessage(partialText("delta", "x"))).toBe(false); + expect(isCompleteModelMessage(approvalDecision("deny", "c"))).toBe(false); + }); +}); + +describe("PartialAggregator", () => { + it("folds a partial_text stream into one complete text message", () => { + const out = aggregateAll([ + partialText("start", "Hel"), + partialText("delta", "lo "), + partialText("delta", "world"), + partialText("stop", "", "completed"), + ]); + expect(out).toHaveLength(1); + expect(isCompleteModelMessage(out[0]!)).toBe(true); + const p = out[0]!.payload as TextPayload; + expect(p.type).toBe("text"); + expect(p.text).toBe("Hello world"); + expect(p.stop_reason).toBe("completed"); + }); + + it("accumulates partial_tool_call arguments and preserves tool_call_id", () => { + const out = aggregateAll([ + partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "c1" }), + partialToolCall({ + eventType: "delta", + name: "exec_command", + arguments: '{"cmd":"ls', + toolCallId: "c1", + }), + partialToolCall({ + eventType: "delta", + name: "exec_command", + arguments: ' -la"}', + toolCallId: "c1", + }), + partialToolCall({ eventType: "stop", name: "exec_command", toolCallId: "c1" }), + ]); + expect(out).toHaveLength(1); + const p = out[0]!.payload as ToolCallPayload; + expect(p.type).toBe("tool_call"); + expect(p.name).toBe("exec_command"); + expect(p.tool_call_id).toBe("c1"); + expect(p.arguments).toBe('{"cmd":"ls -la"}'); + }); + + it("folds tool output image deltas into the complete tool_call_output (拼接 == 完整)", () => { + const dataUrl = "data:image/png;base64,AAAA"; + const out = aggregateAll([ + partialToolCallOutput({ eventType: "start", toolCallId: "c9" }), + partialToolCallOutput({ eventType: "delta", output: "image/png, 4 B", toolCallId: "c9" }), + partialToolCallOutput({ eventType: "delta", toolCallId: "c9", images: [dataUrl] }), + partialToolCallOutput({ eventType: "stop", toolCallId: "c9" }), + ]); + expect(out).toHaveLength(1); + const p = out[0]!.payload as ToolCallOutputPayload; + expect(p.type).toBe("tool_call_output"); + expect(p.output).toBe("image/png, 4 B"); + expect(p.images).toEqual([dataUrl]); + }); + + it("passes through complete and event messages unchanged and keeps order", () => { + const tc = toolCall({ name: "x", arguments: "{}", toolCallId: "c2" }); + const out = aggregateAll([userText("q"), tc, approvalDecision("allow", "c2")]); + expect(out).toHaveLength(3); + // The session_meta payload has no inner type field, so consumers must first narrow by the + // outer type; here we assert order and passthrough with a loose read. + const payloadTypes = out.map((m) => (m.payload as { type?: string }).type); + expect(payloadTypes).toEqual(["text", "tool_call", "approval_decision"]); + expect(out.map((m) => m.type)).toEqual(["model_msg", "model_msg", "event_msg"]); + }); + + it("flush emits unterminated fragments", () => { + const agg = new PartialAggregator(); + expect(agg.push(partialText("start", "abc"))).toEqual([]); + expect(agg.push(partialText("delta", "def"))).toEqual([]); + const flushed = agg.flush(); + expect(flushed).toHaveLength(1); + expect((flushed[0]!.payload as TextPayload).text).toBe("abcdef"); + }); +}); diff --git a/packages/core/test/provider-keys.ts b/packages/core/test/provider-keys.ts new file mode 100644 index 0000000..db70233 --- /dev/null +++ b/packages/core/test/provider-keys.ts @@ -0,0 +1,24 @@ +/** + * Fake API keys for testing: the provider SDK requires a credential at **construction time** + * (throwing "Missing credentials" if absent), and `createSession` constructs an LLM client for + * the default model. + * + * The default model uses the OpenAI protocol (DeepSeek), so `OPENAI_API_KEY` must have a value; + * Anthropic's is stubbed too, for test cases that explicitly specify a claude model. The keys + * are only used to construct the client; tests never actually send a request. CI has no keys + * at all, while most local dev machines do -- without stubbing, tests would "pass locally, + * fail in CI." + */ +const KEYS = ["OPENAI_API_KEY", "ANTHROPIC_API_KEY"] as const; + +/** Stubs in fake keys and returns a restore function (call it in afterEach). */ +export function stubProviderKeys(): () => void { + const prev = KEYS.map((k) => [k, process.env[k]] as const); + for (const k of KEYS) process.env[k] = "test-key-not-used"; + return () => { + for (const [k, v] of prev) { + if (v === undefined) delete process.env[k]; + else process.env[k] = v; + } + }; +} diff --git a/packages/core/test/read-image.test.ts b/packages/core/test/read-image.test.ts new file mode 100644 index 0000000..0d8f1fe --- /dev/null +++ b/packages/core/test/read-image.test.ts @@ -0,0 +1,147 @@ +/** + * Unit tests for the read_image tool (no network): local file reading (relative-path + * resolution / magic-number sniffing / missing file / over the size limit / unsupported type / + * missing argument) and http(s) URL download (stubbing global fetch: success and non-2xx). + * Directly drives BuiltinTool.execute and captures the generator's return value -- see + * environment.test.ts for the Environment-side images assembly. + */ +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { + MAX_IMAGE_BYTES, + READ_IMAGE_NAME, + createReadImageTool, +} from "../src/environment/tools/read-image.js"; +import type { ToolResult } from "../src/environment/tools/types.js"; +import type { OmniMessage } from "../src/omnimessage/index.js"; +import type { ToolDefinitionConfig } from "../src/interfaces.js"; + +/** Full bytes of a 1x1 transparent PNG (including the magic number, enough for mime sniffing + * and data URL assertions). */ +const PNG_1X1 = Buffer.from( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==", + "base64", +); + +const definition: ToolDefinitionConfig = { + name: READ_IMAGE_NAME, + description: "read image", + permission: "r", +}; + +/** Runs one tool execution: collects streamed messages, concatenates text deltas, and captures + * the generator's return value. */ +async function run(args: Record, workspaceDir: string) { + const tool = createReadImageTool(definition); + const gen = tool.execute(args, { workspaceDir, toolCallId: "c1" }); + const messages: OmniMessage[] = []; + let result: ToolResult | void; + for (;;) { + const res = await gen.next(); + if (res.done) { + result = res.value; + break; + } + messages.push(res.value); + } + const text = messages.map((m) => (m.payload as { output?: string }).output ?? "").join(""); + return { messages, result, text }; +} + +let tmp: string; + +beforeEach(async () => { + tmp = await mkdtemp(path.join(tmpdir(), "penguin-readimg-")); +}); + +afterEach(async () => { + vi.unstubAllGlobals(); + await rm(tmp, { recursive: true, force: true }); +}); + +describe("read_image — 本地文件", () => { + it("按相对路径读取 png,输出 data URL 与一行 mime/大小说明", async () => { + await writeFile(path.join(tmp, "img.png"), PNG_1X1); + const { result, text } = await run({ source: "img.png" }, tmp); + expect(result?.stopReason).toBeUndefined(); // Defaults to completed + expect(result?.images).toEqual([`data:image/png;base64,${PNG_1X1.toString("base64")}`]); + expect(text).toBe(`image/png, ${PNG_1X1.length} B`); + }); + + it("扩展名与内容不符时以魔数嗅探为准", async () => { + await writeFile(path.join(tmp, "photo.jpg"), PNG_1X1); // Content is actually PNG + const { result } = await run({ source: "photo.jpg" }, tmp); + expect(result?.images?.[0]).toMatch(/^data:image\/png;base64,/); + }); + + it("文件不存在时以 failed 收尾并输出解释", async () => { + const { result, text } = await run({ source: "missing.png" }, tmp); + expect(result?.stopReason).toBe("failed"); + expect(result?.images).toBeUndefined(); + expect(text).toContain("missing.png"); + }); + + it("超过大小上限时以 failed 收尾", async () => { + const big = Buffer.alloc(MAX_IMAGE_BYTES + 1); + PNG_1X1.copy(big); // Header carries the PNG magic number, ensuring the failure is due to size, not type + await writeFile(path.join(tmp, "big.png"), big); + const { result, text } = await run({ source: "big.png" }, tmp); + expect(result?.stopReason).toBe("failed"); + expect(text).toContain("too large"); + }); + + it("不支持的图片类型以 failed 收尾", async () => { + await writeFile(path.join(tmp, "img.bmp"), Buffer.from("BM not really an image")); + const { result, text } = await run({ source: "img.bmp" }, tmp); + expect(result?.stopReason).toBe("failed"); + expect(text).toContain("Unsupported image type"); + }); + + it("source 指向目录时以 failed 收尾并明确说明(不透传 EISDIR)", async () => { + await mkdir(path.join(tmp, "subdir")); + const { result, text } = await run({ source: "subdir" }, tmp); + expect(result?.stopReason).toBe("failed"); + expect(text).toContain("not a file"); + expect(text).not.toContain("EISDIR"); + }); + + it("空文件以 failed 收尾(扩展名兜底不得放行空 base64)", async () => { + await writeFile(path.join(tmp, "empty.png"), Buffer.alloc(0)); + const { result, text } = await run({ source: "empty.png" }, tmp); + expect(result?.stopReason).toBe("failed"); + expect(text).toContain("empty"); + }); + + it("缺少 source 参数以 failed 收尾", async () => { + const { result, text } = await run({}, tmp); + expect(result?.stopReason).toBe("failed"); + expect(text).toContain('"source"'); + }); +}); + +describe("read_image — http(s) URL", () => { + it("经全局 fetch 下载并取响应头 content-type 作为 mime", async () => { + const fetchMock = vi.fn( + async (_input: unknown) => + new Response(PNG_1X1, { status: 200, headers: { "content-type": "image/png" } }), + ); + vi.stubGlobal("fetch", fetchMock); + const { result, text } = await run({ source: "https://example.com/a" }, tmp); + expect(fetchMock).toHaveBeenCalledTimes(1); + expect(fetchMock.mock.calls[0]![0]).toBe("https://example.com/a"); + expect(result?.images).toEqual([`data:image/png;base64,${PNG_1X1.toString("base64")}`]); + expect(text).toContain("image/png"); + }); + + it("非 2xx 响应以 failed 收尾并带状态码", async () => { + vi.stubGlobal( + "fetch", + vi.fn(async () => new Response("nope", { status: 404 })), + ); + const { result, text } = await run({ source: "https://example.com/missing.png" }, tmp); + expect(result?.stopReason).toBe("failed"); + expect(text).toContain("404"); + }); +}); diff --git a/packages/core/test/replay.test.ts b/packages/core/test/replay.test.ts new file mode 100644 index 0000000..67850c9 --- /dev/null +++ b/packages/core/test/replay.test.ts @@ -0,0 +1,469 @@ +/** + * Trace replay: + * + * - Per-round judgment: completed rounds enter history; uncommitted rounds (not completed / have + * a start but no stop) are dropped entirely, keeping only outputs paired with already-committed + * tool_calls; trailing input is kept as-is as carry-over. + * - Pairing fallback: committed tool_calls with no paired output get an interrupted-state placeholder. + * - Compaction wrap-up (file level): summarize rebuilds , discard leaves no + * pending input; failed compaction rounds are dropped by the generic rule. + * - Tolerates a truncated trailing line left by an abnormal process exit. + * - Round-trip: a Trace written out by the engine, once replayed, matches the history the model actually received. + */ +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { + assistantText, + compactionBegin, + compactionEnd, + emptyTokenCounts, + requestBegin, + requestEnd, + sessionMeta, + thinkingMessage, + tokenUsage, + toolCall, + toolCallOutput, + userText, +} from "../src/omnimessage/index.js"; +import type { OmniMessage, TokenCounts } from "../src/omnimessage/index.js"; +import { parseTraceLines, resumeTrace } from "../src/trace/resume.js"; +import { ContextEngine } from "../src/engine/context-engine.js"; +import { Environment } from "../src/environment/index.js"; +import { Writer, readTrace } from "../src/trace/index.js"; +import type { ApproveFn, LLMInterface } from "../src/interfaces.js"; + +const usage = (total: number): TokenCounts => ({ + cache_read: 0, + cache_write: 0, + output: 1, + total, +}); + +function meta(): OmniMessage { + return sessionMeta({ + session_id: "session-2026-07-06-10-00-00-abcdef01", + provider: "anthropic", + model_id: "claude-sonnet-4-6", + model_context_window: 1000000, + system_prompt: "SP", + tools: [], + thinking_level: "default", + agent_state: "/agent/state", + workspace: "/ws", + }); +} + +function textsOf(msgs: OmniMessage[]): string[] { + return msgs.map((m) => { + const p = m.payload as { + type?: string; + text?: string; + output?: string; + name?: string; + thinking?: string; + }; + return p.text ?? p.output ?? p.thinking ?? p.name ?? p.type ?? ""; + }); +} + +describe("resumeTrace", () => { + it("committed rounds enter history; trailing inputs become carry-over", () => { + const result = resumeTrace([ + meta(), + userText("hello"), + requestBegin(), + assistantText("hi"), + requestEnd("completed"), + tokenUsage(usage(10), usage(10)), + userText("tail input"), // the request never got a chance to start + ]); + expect(textsOf(result.history)).toEqual(["hello", "hi"]); + expect(textsOf(result.carryOver)).toEqual(["tail input"]); + expect(result.sessionTurns).toBe(1); + expect(result.sessionTokens.total).toBe(10); + expect(result.lastRequestTotal).toBe(10); + expect(result.contextClosed).toBe(false); + }); + + it("re-carries the uncommitted round's raw input into the retried round", () => { + // The synthesized carry-over (flatten) is never written to Trace: replay does its best, + // merging the unanswered raw input as-is into the retried round (the history content differs + // in wording from the flatten AgentHub actually received, but matches in structure and information). + const result = resumeTrace([ + meta(), + userText("A"), + requestBegin(), + requestEnd("timeout"), // failed with zero output + requestBegin(), + assistantText("ok"), + requestEnd("completed"), + tokenUsage(usage(20), usage(20)), + ]); + expect(textsOf(result.history)).toEqual(["A", "ok"]); + expect(result.carryOver).toEqual([]); + }); + + it("keeps structured outputs pairing committed tool_calls when a later round fails", () => { + const result = resumeTrace([ + meta(), + userText("run it"), + requestBegin(), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc1" }), + requestEnd("completed"), + tokenUsage(usage(10), usage(10)), + toolCallOutput({ output: "result-1", toolCallId: "tc1" }), + requestBegin(), + thinkingMessage("half", "aborted"), // the replay request was interrupted + requestEnd("aborted"), + ]); + expect(textsOf(result.history)).toEqual(["run it", "exec_command"]); + // tc1 was committed but unanswered: its output is kept pending (structured re-delivery); the half-finished thinking is discarded. + expect(textsOf(result.carryOver)).toEqual(["result-1"]); + }); + + it("treats begin-without-end as uncommitted: raw input re-carried, half-products lost", () => { + const result = resumeTrace([ + meta(), + userText("A"), + requestBegin(), + thinkingMessage("half", "aborted"), + // The process exited during the request: no end. + ]); + expect(result.history).toEqual([]); + // The last unanswered input is resent as-is; the model's half-finished output is allowed to be lost. + expect(textsOf(result.carryOver)).toEqual(["A"]); + // The render view still shows every complete message. + expect(textsOf(result.renderMessages)).toEqual(["A", "half"]); + }); + + it("backfills placeholder outputs for committed tool_calls with no paired output", () => { + const result = resumeTrace([ + meta(), + userText("go"), + requestBegin(), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc1" }), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc2" }), + requestEnd("completed"), + tokenUsage(usage(10), usage(10)), + toolCallOutput({ output: "done-1", toolCallId: "tc1" }), + // tc2's output was lost along with the process. + ]); + // carry-over = real trailing output + in-memory synthesized placeholder (never written to Trace), pairing complete. + expect(result.carryOver).toHaveLength(2); + const backfill = result.carryOver[1]!.payload as { tool_call_id: string; output: string }; + expect(backfill.tool_call_id).toBe("tc2"); + expect(backfill.output).toContain("interrupted"); + expect(textsOf(result.carryOver)).toEqual(["done-1", backfill.output]); + }); + + it("routes user-side messages inside a request span to the next round's input", () => { + const result = resumeTrace([ + meta(), + userText("go"), + requestBegin(), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc1" }), + // A parallel tool finished during the request: its output lands between start and stop. + toolCallOutput({ output: "early", toolCallId: "tc1" }), + requestEnd("completed"), + tokenUsage(usage(10), usage(10)), + ]); + expect(textsOf(result.history)).toEqual(["go", "exec_command"]); + expect(textsOf(result.carryOver)).toEqual(["early"]); + }); + + it("closed context (summarize): empty history, summary rebuilt from compaction output", () => { + const result = resumeTrace([ + meta(), + userText("hello"), + requestBegin(), + assistantText("hi"), + requestEnd("completed"), + tokenUsage(usage(10), usage(10)), + compactionBegin({ reason: "context", mode: "summarize", context: 10, turns: 1 }), + userText("please summarize"), + requestBegin(), + assistantText("the gist"), + requestEnd("completed"), + tokenUsage(usage(20), usage(20)), + compactionEnd({ reason: "context", mode: "summarize", status: "completed" }), + ]); + expect(result.contextClosed).toBe(true); + expect(result.history).toEqual([]); + expect(result.renderMessages).toEqual([]); + const summary = result.pendingSummary!.payload as { text: string }; + expect(summary.text).toBe("\nthe gist\n"); + expect(result.sessionTurns).toBe(0); + expect(result.sessionTokens.total).toBe(20); // Token carry-over includes compaction consumption + }); + + it("closed context (discard): empty history and no pending summary", () => { + const result = resumeTrace([ + meta(), + userText("hello"), + requestBegin(), + assistantText("hi"), + requestEnd("completed"), + tokenUsage(usage(10), usage(10)), + compactionBegin({ reason: "manual", mode: "discard", context: 10, turns: 1 }), + compactionEnd({ reason: "manual", mode: "discard", status: "completed" }), + ]); + expect(result.contextClosed).toBe(true); + expect(result.history).toEqual([]); + expect(result.pendingSummary).toBeUndefined(); + }); + + it("drops failed compaction rounds via the generic rule (prompt not in history)", () => { + const result = resumeTrace([ + meta(), + userText("hello"), + requestBegin(), + assistantText("hi"), + requestEnd("completed"), + tokenUsage(usage(10), usage(10)), + compactionBegin({ reason: "context", mode: "summarize", context: 10, turns: 1 }), + userText("please summarize"), + requestBegin(), + requestEnd("failed"), + compactionEnd({ reason: "context", mode: "summarize", status: "failed" }), + userText("continue"), + requestBegin(), + assistantText("sure"), + requestEnd("completed"), + tokenUsage(usage(30), usage(30)), + ]); + expect(result.contextClosed).toBe(false); + expect(textsOf(result.history)).toEqual(["hello", "hi", "continue", "sure"]); + expect(result.carryOver).toEqual([]); + expect(result.sessionTurns).toBe(2); + }); +}); + +describe("parseTraceLines", () => { + it("tolerates a torn trailing line (process crash mid-write)", () => { + const content = `${JSON.stringify(userText("a"))}\n{"timestamp":"2026-07-06T`; + const msgs = parseTraceLines(content); + expect(msgs).toHaveLength(1); + }); + + it("throws on mid-file corruption", () => { + const content = `not-json\n${JSON.stringify(userText("a"))}\n`; + expect(() => parseTraceLines(content)).toThrow(); + }); +}); + +describe("engine trace round-trip", () => { + let workspace: string; + let traces: string; + + beforeEach(async () => { + workspace = await mkdtemp(join(tmpdir(), "penguin-replay-ws-")); + traces = await mkdtemp(join(tmpdir(), "penguin-replay-tr-")); + }); + + afterEach(async () => { + await rm(workspace, { recursive: true, force: true }); + await rm(traces, { recursive: true, force: true }); + }); + + it("replaying an engine-written trace reconstructs the committed history", async () => { + // Two rounds of dialogue: the first round issues and executes a tool call, the second round + // wraps up -- the engine writes Trace (including request events); replay should reconstruct + // history matching what was actually committed to AgentHub, with no leftover carry-over. + let call = 0; + const llm: LLMInterface = { + async *streamGenerate() { + call += 1; + if (call === 1) { + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "printf ok" }), + toolCallId: "rt1", + }); + yield tokenUsage(usage(10), usage(10)); + return { status: "completed" }; + } + yield assistantText("done"); + yield tokenUsage(usage(20), usage(20)); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: { + customTools: [ + { + name: "exec_command", + description: "Run a shell command.", + parameters: { type: "object", properties: { cmd: { type: "string" } } }, + permission: "rw" as const, + }, + ], + mcpServers: [], + }, + }); + const trace = new Writer({ tracesDir: traces, sessionId: "sess-roundtrip" }); + const engine = new ContextEngine({ llm, environment, trace }); + const allow: ApproveFn = async () => "allow"; + for await (const _ of engine.run([userText("go")], { approve: allow })) { + // consume + } + + const recorded = await readTrace(trace.currentPath()); + // request boundary events are written in pairs: one pair per round, two rounds total. + const requestEvents = recorded.filter((m) => + ((m.payload as { type?: string }).type ?? "").startsWith("request_"), + ); + expect(requestEvents.map((m) => (m.payload as { type?: string }).type)).toEqual([ + "request_begin", + "request_end", + "request_begin", + "request_end", + ]); + + const result = resumeTrace(recorded); + expect(result.carryOver).toEqual([]); + expect(result.sessionTurns).toBe(2); + // History = input -> tool_call -> tool output -> final reply, in the same order as committed. + const kinds = result.history.map((m) => (m.payload as { type?: string }).type); + expect(kinds).toEqual(["text", "tool_call", "tool_call_output", "text"]); + }); + + it("aborted run leaves a replayable trace: the raw input is re-carried (flatten not persisted)", async () => { + let call = 0; + const llm: LLMInterface = { + async *streamGenerate() { + call += 1; + if (call === 1) { + yield thinkingMessage("half", "aborted"); + return { status: "aborted" }; + } + yield assistantText("ok"); + yield tokenUsage(usage(5), usage(5)); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: { customTools: [], mcpServers: [] }, + }); + const trace = new Writer({ tracesDir: traces, sessionId: "sess-abort" }); + const engine = new ContextEngine({ llm, environment, trace }); + for await (const _ of engine.run([userText("go")], { approve: async () => "deny" })) { + // consume + } + + // If the process exits here: Trace has no synthesized flatten; replay treats the raw input as + // pending input as-is, and the original round never enters history. + const recorded = await readTrace(trace.currentPath()); + expect( + recorded.some((m) => + ((m.payload as { text?: string }).text ?? "").includes(""), + ), + ).toBe(false); + const result = resumeTrace(recorded); + expect(result.history).toEqual([]); + expect(textsOf(result.carryOver)).toEqual(["go"]); + }); +}); + +describe("resumeTrace regressions (PR #39 review)", () => { + it("drops the aborted round's own tool outputs regardless of landing before or after the end", () => { + // Tools and the LLM stream run concurrently: an orphaned output may land on disk before or + // after that round's end, and neither may be re-delivered as a structured result (its + // tool_call isn't in history, so re-delivering it would produce an orphan tool_result with + // no preceding tool_use); the raw input is resent as-is. + const result = resumeTrace([ + meta(), + userText("go"), + requestBegin(), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc1" }), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc2" }), + toolCallOutput({ output: "early", toolCallId: "tc1" }), + requestEnd("aborted"), + toolCallOutput({ output: "late", toolCallId: "tc2", stopReason: "aborted" }), + ]); + expect(result.history).toEqual([]); + expect(textsOf(result.carryOver)).toEqual(["go"]); + }); + + it("filters the dropped round's tool outputs out of the next committed round's input snapshot", () => { + // Main reconnect flow: a tool executes during a timed-out attempt (its output lands on disk), + // and the retry round completes. The dropped round's tool_call is not in history -- when its + // output is snapshotted into the retry round's input it must be filtered out, otherwise the + // history injected via setHistory contains an orphan tool_result with no preceding tool_use, + // and every request after resume gets rejected by the provider (400). + const result = resumeTrace([ + meta(), + userText("go"), + requestBegin(), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc1" }), + toolCallOutput({ output: "ran-during-timeout", toolCallId: "tc1" }), + requestEnd("timeout"), // this round is dropped: tc1 never entered AgentHub history + requestBegin(), + assistantText("recovered"), + requestEnd("completed"), + tokenUsage(usage(20), usage(20)), + ]); + // History = original input + retry round output; the orphaned tool_call_output neither enters history nor is re-delivered. + expect(textsOf(result.history)).toEqual(["go", "recovered"]); + expect( + result.history.some((m) => (m.payload as { type?: string }).type === "tool_call_output"), + ).toBe(false); + expect(result.carryOver).toEqual([]); + }); + + it("repairs history structure with in-memory placeholders for previously unpaired committed calls", () => { + // Scenario: the placeholder synthesized on a previous resume was sent out with the request but + // never written to Trace; a subsequent committed round in Trace is therefore missing that + // pairing. Replay must re-synthesize the placeholder and inject it into history before that + // round's input, guaranteeing every assistant tool_use is followed by a tool_result. + const result = resumeTrace([ + meta(), + userText("go"), + requestBegin(), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc1" }), + requestEnd("completed"), + tokenUsage(usage(10), usage(10)), + // (process exits -> resume -> placeholder sent out with the next round but never persisted) + userText("next"), + requestBegin(), + assistantText("ok"), + requestEnd("completed"), + tokenUsage(usage(20), usage(20)), + ]); + const kinds = result.history.map((m) => (m.payload as { type?: string }).type); + expect(kinds).toEqual(["text", "tool_call", "tool_call_output", "text", "text"]); + const repaired = result.history[2]!.payload as { tool_call_id: string; output: string }; + expect(repaired.tool_call_id).toBe("tc1"); + expect(repaired.output).toContain("interrupted"); + expect(result.carryOver).toEqual([]); + }); + + it("closed context (summarize) with a textless compaction output yields an empty summary", () => { + // The compaction request completed but produced no text (e.g. thinking-only): the summary is + // empty, and must not fall back to an earlier round's ordinary answer (consistent with the + // in-process extractSummary("") behavior). + const result = resumeTrace([ + meta(), + userText("hello"), + requestBegin(), + assistantText("The answer is 42."), + requestEnd("completed"), + tokenUsage(usage(10), usage(10)), + compactionBegin({ reason: "context", mode: "summarize", context: 10, turns: 1 }), + userText("please summarize"), + requestBegin(), + thinkingMessage("thinking only, no text"), + requestEnd("completed"), + tokenUsage(usage(20), usage(20)), + compactionEnd({ reason: "context", mode: "summarize", status: "completed" }), + ]); + expect(result.contextClosed).toBe(true); + const summary = result.pendingSummary!.payload as { text: string }; + expect(summary.text).toBe("\n\n"); + expect(summary.text).not.toContain("42"); + }); +}); diff --git a/packages/core/test/resume.test.ts b/packages/core/test/resume.test.ts new file mode 100644 index 0000000..99ee637 --- /dev/null +++ b/packages/core/test/resume.test.ts @@ -0,0 +1,288 @@ +/** + * Session resume: `agent.resumeSession` and setHistory injection. + * + * - Resume source is the Trace file with the latest index; config carries over from session_meta + * (Workspace / Model cannot be swapped). + * - Pairing-fallback placeholders, once constructed, are written into the original trace file; + * session_meta is never written twice. + * - Errors when the session doesn't exist / the workspace is missing / the model is no longer in the project config. + * - `groupHistoryToUniMessages` groups by adjacent same role; `GenerativeModel.setHistory` injects into AgentHub. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { createAgent } from "../src/index.js"; +import { + abortEvent, + assistantText, + requestBegin, + requestEnd, + sessionMeta, + tokenUsage, + toolCall, + userText, +} from "../src/omnimessage/index.js"; +import type { OmniMessage, TokenCounts } from "../src/omnimessage/index.js"; +import { GenerativeModel, groupHistoryToUniMessages } from "../src/llm/index.js"; +import { readTrace } from "../src/trace/index.js"; +import { tracesDir } from "../src/state/paths.js"; +import { stubProviderKeys } from "./provider-keys.js"; + +// The default project config ships with this model ((provider, model_id) pair reference; model_id is the upstream id). +const MODEL = { provider: "anthropic", model_id: "claude-sonnet-4-6" }; + +let tmpRoot: string; +let workspace: string; +let prevHome: string | undefined; +let restoreKeys: () => void; + +beforeEach(async () => { + prevHome = process.env.PENGUIN_HOME; + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-resume-")); + workspace = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-resume-ws-")); + process.env.PENGUIN_HOME = tmpRoot; + restoreKeys = stubProviderKeys(); +}); + +afterEach(async () => { + if (prevHome === undefined) delete process.env.PENGUIN_HOME; + else process.env.PENGUIN_HOME = prevHome; + restoreKeys(); + await fs.rm(tmpRoot, { recursive: true, force: true }); + await fs.rm(workspace, { recursive: true, force: true }); +}); + +const usage = (total: number): TokenCounts => ({ + cache_read: 0, + cache_write: 0, + output: 1, + total, +}); + +/** Manually constructs a session's trace file (simulating a record left behind by a previous process). */ +async function writeTraceFile( + root: string, + sessionId: string, + messages: OmniMessage[], + opts?: { dateDir?: string; index?: string }, +): Promise { + const dir = path.join( + tracesDir(root, "default_project", "default_agent"), + opts?.dateDir ?? "2026-07-06", + ); + await fs.mkdir(dir, { recursive: true }); + const file = path.join(dir, `${sessionId}_${opts?.index ?? "001"}.jsonl`); + await fs.writeFile(file, messages.map((m) => JSON.stringify(m)).join("\n") + "\n", "utf8"); + return file; +} + +function metaFor(sessionId: string, workspaceDir: string, model = MODEL): OmniMessage { + return sessionMeta({ + session_id: sessionId, + provider: model.provider, + model_id: model.model_id, + model_context_window: 1000000, + system_prompt: "ORIGINAL SYSTEM PROMPT", + tools: [], + thinking_level: "default", + agent_state: "/agent/state", + workspace: workspaceDir, + }); +} + +describe("agent.resumeSession", () => { + const SID = "session-2026-07-06-10-00-00-abcdef01"; + + it("resumes from the latest trace file and exposes render history", async () => { + const agent = await createAgent({}); + await writeTraceFile(tmpRoot, SID, [ + metaFor(SID, workspace), + userText("hello"), + requestBegin(), + assistantText("hi there"), + requestEnd("completed"), + tokenUsage(usage(42), usage(42)), + ]); + + const session = await agent.resumeSession({ sessionId: SID }); + expect(session.sessionId).toBe(SID); + expect(session.provider).toBe(MODEL.provider); + expect(session.modelId).toBe(MODEL.model_id); + expect(session.workspaceDir).toBe(workspace); + const texts = (session.resumedHistory ?? []).map( + (m) => (m.payload as { text?: string }).text ?? "", + ); + expect(texts).toEqual(["hello", "hi there"]); + }); + + it("keeps abort events in resumed render history", async () => { + const agent = await createAgent({}); + await writeTraceFile(tmpRoot, SID, [ + metaFor(SID, workspace), + userText("long task"), + requestBegin(), + assistantText("partial answer", "aborted"), + requestEnd("aborted"), + abortEvent("aborted by user"), + ]); + + const session = await agent.resumeSession({ sessionId: SID }); + expect( + (session.resumedHistory ?? []).map((m) => (m.payload as { type?: string }).type), + ).toEqual(["text", "text", "abort"]); + }); + + it("does not write pairing placeholders to the trace file (resume is side-effect free)", async () => { + const agent = await createAgent({}); + const file = await writeTraceFile(tmpRoot, SID, [ + metaFor(SID, workspace), + userText("go"), + requestBegin(), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc1" }), + requestEnd("completed"), + tokenUsage(usage(10), usage(10)), + // tc1's output was lost along with the process: the pairing placeholder is synthesized in memory and sent out with the next run, never persisted. + ]); + const before = await readTrace(file); + + await agent.resumeSession({ sessionId: SID }); + await agent.resumeSession({ sessionId: SID }); // resuming again has no side effects either + const after = await readTrace(file); + expect(after).toHaveLength(before.length); + expect( + after.filter((m) => + ((m.payload as { output?: string }).output ?? "").includes("interrupted"), + ), + ).toHaveLength(0); + // session_meta is never written twice. + expect(after.filter((m) => m.type === "session_meta")).toHaveLength(1); + }); + + it("picks the latest index when the context was compacted into multiple files", async () => { + const agent = await createAgent({}); + await writeTraceFile(tmpRoot, SID, [metaFor(SID, workspace), userText("old context")], { + index: "001", + }); + await writeTraceFile( + tmpRoot, + SID, + [ + metaFor(SID, workspace), + userText("gist"), + requestBegin(), + assistantText("resumed context"), + requestEnd("completed"), + tokenUsage(usage(5), usage(5)), + ], + { index: "002" }, + ); + + const session = await agent.resumeSession({ sessionId: SID }); + const texts = (session.resumedHistory ?? []).map( + (m) => (m.payload as { text?: string }).text ?? "", + ); + expect(texts).toEqual(["gist", "resumed context"]); + }); + + it("errors when the session does not exist", async () => { + const agent = await createAgent({}); + await expect(agent.resumeSession({ sessionId: "session-none" })).rejects.toThrow( + /Session 不存在/, + ); + }); + + it("errors when the recorded workspace no longer exists (PRN-004: no auto-create)", async () => { + const agent = await createAgent({}); + const gone = path.join(workspace, "gone"); + await writeTraceFile(tmpRoot, SID, [metaFor(SID, gone), userText("x")]); + await expect(agent.resumeSession({ sessionId: SID })).rejects.toThrow(/Workspace 已不存在/); + }); + + it("errors when the recorded model is no longer in the project config", async () => { + const agent = await createAgent({}); + await writeTraceFile(tmpRoot, SID, [ + metaFor(SID, workspace, { provider: "custom", model_id: "vanished-model" }), + userText("x"), + ]); + await expect(agent.resumeSession({ sessionId: SID })).rejects.toThrow(/不在 Project 配置/); + }); + + it("errors clearly when session_meta lacks provider (old-format trace, no migration)", async () => { + // An old-format trace's session_meta only has model_id (from the composite-id era): no backward compat, just a clear error. + const agent = await createAgent({}); + const legacy = metaFor(SID, workspace); + delete (legacy.payload as { provider?: string }).provider; + await writeTraceFile(tmpRoot, SID, [legacy, userText("x")]); + await expect(agent.resumeSession({ sessionId: SID })).rejects.toThrow(/旧版本/); + }); + + it("latestSessionId returns the newest session by embedded timestamp", async () => { + const agent = await createAgent({}); + expect(await agent.latestSessionId()).toBeNull(); + const older = "session-2026-07-05-09-00-00-aaaaaaaa"; + const newer = "session-2026-07-06-11-00-00-bbbbbbbb"; + await writeTraceFile(tmpRoot, older, [metaFor(older, workspace), userText("older")], { + dateDir: "2026-07-05", + }); + await writeTraceFile(tmpRoot, newer, [metaFor(newer, workspace), userText("newer")], { + dateDir: "2026-07-06", + }); + expect(await agent.latestSessionId()).toBe(newer); + }); + + it("latestSessionId ignores empty traces that only contain session_meta", async () => { + const agent = await createAgent({}); + const older = "session-2026-07-05-09-00-00-aaaaaaaa"; + const emptyNewer = "session-2026-07-06-11-00-00-bbbbbbbb"; + await writeTraceFile(tmpRoot, older, [metaFor(older, workspace), userText("older")], { + dateDir: "2026-07-05", + }); + await writeTraceFile(tmpRoot, emptyNewer, [metaFor(emptyNewer, workspace)], { + dateDir: "2026-07-06", + }); + expect(await agent.latestSessionId()).toBe(older); + }); + + it("manual compact on a new empty session does not create a resumable trace", async () => { + const agent = await createAgent({}); + const session = await agent.createSession({ workspaceDir: workspace }); + const messages = []; + for await (const msg of session.compact()) messages.push(msg); + expect(messages).toHaveLength(0); + expect(await agent.latestSessionId()).toBeNull(); + }); +}); + +describe("setHistory injection", () => { + it("groupHistoryToUniMessages groups adjacent same-role messages into UniMessages", () => { + const uni = groupHistoryToUniMessages([ + userText("hello"), + assistantText("hi"), + toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "tc1" }), + { + ...userText("ignored-shape"), + payload: { + type: "tool_call_output", + role: "user", + output: "out", + tool_call_id: "tc1", + stop_reason: "completed", + }, + } as OmniMessage, + userText("next"), + assistantText("done"), + ]); + expect(uni.map((m) => m.role)).toEqual(["user", "assistant", "user", "assistant"]); + expect(uni[1]!.content_items.map((c) => c.type)).toEqual(["text", "tool_call"]); + expect(uni[2]!.content_items.map((c) => c.type)).toEqual(["tool_result", "text"]); + }); + + it("GenerativeModel.setHistory seeds the AgentHub client history", () => { + // GenerativeModel takes the request id sent to AgentHub (the upstream id), not the storage id. + const model = new GenerativeModel({ modelId: "claude-sonnet-4-6", tools: [] }); + model.setHistory([userText("hello"), assistantText("hi")]); + const client = (model as unknown as { client: { getHistory(): unknown[] } }).client; + expect(client.getHistory()).toHaveLength(2); + }); +}); diff --git a/packages/core/test/session-title.test.ts b/packages/core/test/session-title.test.ts new file mode 100644 index 0000000..823686e --- /dev/null +++ b/packages/core/test/session-title.test.ts @@ -0,0 +1,175 @@ +/** + * Session title generation unit tests: prompt shape, sanitization rules, single-shot request + * driving (fake LLM), and Session.generateTitle's composition-layer wiring (no real requests sent). + */ +import { describe, it, expect } from "vitest"; +import { + assistantText, + buildTitlePrompt, + emptyTokenCounts, + generateTitleWithLLM, + sanitizeTitle, + Session, + thinkingMessage, + tokenUsage, + userText, +} from "../src/index.js"; +import type { + EnvironmentInterface, + LLMInterface, + LLMOutcome, + OmniMessage, + SessionMetaPayload, +} from "../src/index.js"; + +/** A fake LLM: yields the given messages and finishes with the outcome; records the prompt received. */ +function fakeLLM( + outputs: OmniMessage[], + outcome: LLMOutcome = { status: "completed" }, + seenPrompts: string[] = [], +): LLMInterface { + return { + async *streamGenerate({ newMessages }) { + const first = newMessages[0]; + if (first) seenPrompts.push((first.payload as { text: string }).text); + for (const msg of outputs) yield msg; + return outcome; + }, + }; +} + +const fakeEnvironment: EnvironmentInterface = { + listTools: async () => [], + // eslint-disable-next-line require-yield + executeTool: async function* () { + throw new Error("not used"); + }, + toolPermission: () => undefined, +}; + +const META: SessionMetaPayload = { + session_id: "session-title-1", + provider: "custom", + model_id: "m1", + model_context_window: 1000, + system_prompt: "sp", + tools: [], + thinking_level: "default", + agent_state: "/tmp/state", + workspace: "/tmp/w", +}; + +describe("session-title", () => { + it("generateTitleWithLLM:收集模型 text 与用量,清洗后返回", async () => { + const seen: string[] = []; + const result = await generateTitleWithLLM( + fakeLLM( + [ + thinkingMessage("想一下"), // thinking does not count + assistantText("「Tailwind 主题配置」。"), + tokenUsage(emptyTokenCounts(), { cache_read: 1, cache_write: 2, output: 3, total: 6 }), + ], + { status: "completed" }, + seen, + ), + { userText: "解释 @theme", assistantText: "好的……" }, + ); + expect(result.title).toBe("Tailwind 主题配置"); + expect(result.usage).toEqual({ cache_read: 1, cache_write: 2, output: 3, total: 6 }); + expect(seen[0]).toBe(buildTitlePrompt("解释 @theme", "好的……")); + expect(seen[0]).toContain("SAME language"); + }); + + it("素材为空不发请求;outcome 非 completed 时 title 为 null(usage 保留)", async () => { + const seen: string[] = []; + const empty = await generateTitleWithLLM(fakeLLM([], { status: "completed" }, seen), { + userText: " ", + assistantText: "a", + }); + expect(empty).toEqual({ title: null, usage: null }); + expect(seen).toHaveLength(0); + + const failed = await generateTitleWithLLM( + fakeLLM( + [ + assistantText("半截"), + tokenUsage(emptyTokenCounts(), { cache_read: 0, cache_write: 0, output: 1, total: 1 }), + ], + { status: "failed", message: "401" }, + ), + { userText: "u", assistantText: "a" }, + ); + expect(failed.title).toBeNull(); + expect(failed.usage?.total).toBe(1); + }); + + it("助手素材为空也生成(纯工具轮次):只据用户请求,prompt 省去助手段", async () => { + const seen: string[] = []; + const result = await generateTitleWithLLM( + fakeLLM([assistantText("配置 Tailwind 主题")], { status: "completed" }, seen), + { userText: "帮我配置 @theme", assistantText: "" }, + ); + expect(result.title).toBe("配置 Tailwind 主题"); + expect(seen[0]).toBe(buildTitlePrompt("帮我配置 @theme", "")); + expect(seen[0]).not.toContain("[Assistant]"); + }); + + it("sanitizeTitle:剥引号与句读到稳定、折叠空白、超长截断、空返回 null", () => { + expect(sanitizeTitle("“ 构建配置 说明 。”")).toBe("构建配置 说明"); + expect(sanitizeTitle("『标题』!")).toBe("标题"); + expect(sanitizeTitle(" \n ")).toBeNull(); + expect(sanitizeTitle("x".repeat(50))).toHaveLength(30); + }); + + it("Session.generateTitle:经 createBareLLM 发起;未提供工厂时返回 null", async () => { + const withFactory = new Session({ + meta: META, + llm: fakeLLM([]), + environment: fakeEnvironment, + createBareLLM: () => fakeLLM([assistantText("标题 A")]), + }); + expect( + await withFactory.generateTitle({ material: { userText: "u", assistantText: "a" } }), + ).toEqual({ + title: "标题 A", + usage: null, + }); + + const withoutFactory = new Session({ + meta: META, + llm: fakeLLM([]), + environment: fakeEnvironment, + }); + expect(await withoutFactory.generateTitle()).toEqual({ + title: null, + usage: null, + }); + }); + + it("Session.generateTitle:素材自采(run 收集用户输入与模型正文),无需调用方提供", async () => { + const seen: string[] = []; + const session = new Session({ + meta: META, + llm: fakeLLM([thinkingMessage("想想"), assistantText("答案正文")]), + environment: fakeEnvironment, + createBareLLM: () => fakeLLM([assistantText("标题 B")], { status: "completed" }, seen), + }); + for await (const _ of session.run([userText("用户问题")])) { + void _; // Drains the output stream; once run finishes, the material is settled + } + const res = await session.generateTitle(); + expect(res.title).toBe("标题 B"); + // Material = the first Task's user text + model text (thinking does not count), matching + // buildTitlePrompt's shape. + expect(seen[0]).toBe(buildTitlePrompt("用户问题", "答案正文")); + + // No request is sent when no material has been collected (run was never called). + const idle = new Session({ + meta: META, + llm: fakeLLM([]), + environment: fakeEnvironment, + createBareLLM: () => fakeLLM([assistantText("不应产生")]), + }); + expect(await idle.generateTitle()).toEqual({ title: null, usage: null }); + }); +}); diff --git a/packages/core/test/state.test.ts b/packages/core/test/state.test.ts new file mode 100644 index 0000000..e356995 --- /dev/null +++ b/packages/core/test/state.test.ts @@ -0,0 +1,988 @@ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { + AGENT_ID_PLACEHOLDER, + AGENTS_MD_PLACEHOLDER, + VAULT_KEYS_PLACEHOLDER, + SKILL_METADATA_PLACEHOLDER, + CWD_PLACEHOLDER, + DATE_PLACEHOLDER, + DEFAULT_AGENT_ID, + DEFAULT_PROJECT_ID, + MODEL_CATALOG, + OS_VERSION_PLACEHOLDER, + PLATFORM_PLACEHOLDER, + PROJECT_DIR_PLACEHOLDER, + SESSION_ID_PLACEHOLDER, + addModel, + setVisionModel, + agentsMdPath, + agentStateDir, + agentVaultPath, + loadAgentVault, + assembleSystemPrompt, + buildToolConfig, + selectBuiltinToolsForModel, + defaultProjectConfig, + getModel, + isValidVaultKey, + loadOrInitAgentState, + loadProjectConfig, + memoryDir, + scratchpadDir, + projectConfigPath, + removeVaultEntry, + resolveModelRef, + resolveRoot, + setDefaultModel, + setVaultEntry, + skillsDir, + systemConfigPath, + toolsDir, + type ProjectConfig, +} from "../src/state/index.js"; +import { sessionEnvironment } from "../src/internal/session-support.js"; + +let tmpRoot: string; +let prevHome: string | undefined; + +beforeEach(async () => { + prevHome = process.env.PENGUIN_HOME; + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-state-")); + process.env.PENGUIN_HOME = tmpRoot; +}); + +afterEach(async () => { + if (prevHome === undefined) { + delete process.env.PENGUIN_HOME; + } else { + process.env.PENGUIN_HOME = prevHome; + } + await fs.rm(tmpRoot, { recursive: true, force: true }); +}); + +async function exists(p: string): Promise { + try { + await fs.access(p); + return true; + } catch { + return false; + } +} + +describe("paths / resolveRoot", () => { + it("honors PENGUIN_HOME", () => { + expect(resolveRoot()).toBe(tmpRoot); + }); +}); + +describe("loadOrInitAgentState", () => { + it("initializes an empty agent directory with the full state layout", async () => { + const state = await loadOrInitAgentState(); + expect(state.root).toBe(tmpRoot); + expect(state.projectId).toBe(DEFAULT_PROJECT_ID); + expect(state.agentId).toBe(DEFAULT_AGENT_ID); + + const root = tmpRoot; + expect(await exists(systemConfigPath(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true); + expect(await exists(agentsMdPath(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true); + expect(await exists(toolsDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true); + expect(await exists(memoryDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true); + expect(await exists(skillsDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true); + // The scratchpad/ directory alongside agent_state (model temp files get a subdirectory per Session id). + expect(await exists(scratchpadDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true); + + expect(state.stateDir).toBe(agentStateDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID)); + + // The default system Prompt states the Agent's identity, without repeating tool details + // already in the tool schema (Suggested workflows only points to the run_subagent + // delegation entry point). + expect(state.systemConfig.system_prompt).toContain("PenguinHarness"); + expect(state.systemConfig.system_prompt).not.toContain("exec_command"); + // Suggested workflows absorbs Subagent delegation and task conventions (self-reported + // identity as a soft convention, parallelism, file exchange). + expect(state.systemConfig.system_prompt).toContain("# Suggested workflows"); + expect(state.systemConfig.system_prompt).toContain("run_subagent"); + expect(state.systemConfig.system_prompt).toContain("Caller agent"); + // The default AGENTS.md is empty: it carries no preset guidance. + expect(state.agentsMd).toBe(""); + expect(state.systemConfig.system_prompt).toContain(AGENTS_MD_PLACEHOLDER); + expect(state.systemConfig.system_prompt).toContain(SESSION_ID_PLACEHOLDER); + expect(state.systemConfig.system_prompt).toContain(CWD_PLACEHOLDER); + expect(state.systemConfig.system_prompt).toContain(PLATFORM_PLACEHOLDER); + expect(state.systemConfig.system_prompt).toContain(OS_VERSION_PLACEHOLDER); + expect(state.systemConfig.system_prompt).toContain(DATE_PLACEHOLDER); + // AGENTS.md and the Environment injection sit at the end of the template, with AGENTS.md + // before Environment; the wrapper text is written directly into + // the template (the Prompt is transparent about the config). + expect(state.systemConfig.system_prompt).toContain(""); + expect(state.systemConfig.system_prompt).toContain(""); + // The default template explains the semantics of system-synthesized markers to the model, + // and recommends preferring tool use. + expect(state.systemConfig.system_prompt).toContain(""); + expect(state.systemConfig.system_prompt).toContain(""); + expect(state.systemConfig.system_prompt).toContain(""); + expect(state.systemConfig.system_prompt).toContain("# Tool use"); + // Privacy hardening: explicitly forbids reading .project_config.toml (the sole config file, + // which holds API keys) and each Agent's .vault.toml, and states that config can only be + // changed via the CLI (penguin config ...). + expect(state.systemConfig.system_prompt).toContain("Never read"); + expect(state.systemConfig.system_prompt).toContain(".project_config.toml"); + expect(state.systemConfig.system_prompt).toContain("agent_state/.vault.toml"); + expect(state.systemConfig.system_prompt).toContain("CLI-only"); + expect(state.systemConfig.system_prompt).toContain("penguin config"); + expect(state.systemConfig.system_prompt).not.toContain(".credentials.toml"); + expect(state.systemConfig.system_prompt.indexOf(AGENTS_MD_PLACEHOLDER)).toBeLessThan( + state.systemConfig.system_prompt.indexOf("# Environment"), + ); + // The # Vault and # Skills body sections plus their placeholders: the default template + // places them after and before # Environment, in the order + // Vault -> Skills (the statement text is part of the template body, kept even with no + // keys/skills). + const tpl = state.systemConfig.system_prompt; + expect(tpl).toContain("# Vault"); + expect(tpl).toContain(VAULT_KEYS_PLACEHOLDER); + expect(tpl).toContain("# Skills"); + expect(tpl).toContain(SKILL_METADATA_PLACEHOLDER); + expect(tpl).toContain(""); + expect(tpl.indexOf("")).toBeLessThan(tpl.indexOf("# Vault")); + expect(tpl.indexOf("# Vault")).toBeLessThan(tpl.indexOf(VAULT_KEYS_PLACEHOLDER)); + expect(tpl.indexOf(VAULT_KEYS_PLACEHOLDER)).toBeLessThan(tpl.indexOf("# Skills")); + expect(tpl.indexOf("# Skills")).toBeLessThan(tpl.indexOf(SKILL_METADATA_PLACEHOLDER)); + expect(tpl.indexOf(SKILL_METADATA_PLACEHOLDER)).toBeLessThan(tpl.indexOf("# Environment")); + expect(state.systemConfig.model?.max_tokens).toBe(32000); + expect(state.systemConfig.model?.thinking_level).toBe("medium"); + expect(state.systemConfig.model?.timeoutMs).toBe(120000); + expect(state.systemConfig.tools?.mcpServers).toEqual([]); + expect(Object.hasOwn(state.systemConfig, "description")).toBe(false); + expect(Object.hasOwn(state.systemConfig, "subagents")).toBe(false); + }); + + it("loads an existing agent directory and returns the same system prompt", async () => { + const first = await loadOrInitAgentState(); + const second = await loadOrInitAgentState(); + expect(second.systemConfig.system_prompt).toBe(first.systemConfig.system_prompt); + expect(second.systemConfig.system_prompt).toContain("PenguinHarness"); + expect(second.agentsMd).toBe(first.agentsMd); + // The tool config is fully preserved on the load path. + expect(second.systemConfig.tools?.builtin?.[0]?.name).toBe("exec_command"); + }); + + it("respects custom agentId / projectId", async () => { + const state = await loadOrInitAgentState({ agentId: "agent_x", projectId: "proj_y" }); + expect(state.agentId).toBe("agent_x"); + expect(state.projectId).toBe("proj_y"); + expect(await exists(systemConfigPath(tmpRoot, "proj_y", "agent_x"))).toBe(true); + }); +}); + +describe("buildToolConfig", () => { + it("exposes exec/input command, run/input subagent (rw) and read_image (r)", async () => { + const state = await loadOrInitAgentState(); + const cfg = buildToolConfig(state); + expect(cfg.mcpServers).toEqual([]); + expect(cfg.customTools.map((t) => t.name)).toEqual([ + "exec_command", + "input_command", + "run_subagent", + "input_subagent", + "read_image", + "describe_image", + ]); + const exec = cfg.customTools.find((t) => t.name === "exec_command")!; + expect(exec.permission).toBe("rw"); + expect(exec.timeoutMs).toBe(120000); + expect(exec.maxOutputLength).toBe(16000); + expect((exec.parameters as { required?: string[] }).required).toEqual(["cmd"]); + const write = cfg.customTools.find((t) => t.name === "input_command")!; + expect(write.permission).toBe("rw"); + expect((write.parameters as { required?: string[] }).required).toEqual(["process_id"]); + const sub = cfg.customTools.find((t) => t.name === "run_subagent")!; + expect(sub.permission).toBe("rw"); + expect((sub.parameters as { required?: string[] }).required).toEqual(["prompt"]); + const writeSub = cfg.customTools.find((t) => t.name === "input_subagent")!; + expect(writeSub.permission).toBe("rw"); + expect((writeSub.parameters as { required?: string[] }).required).toEqual(["subagent_id"]); + // Both image-reading tool entries are explicitly in the config, each declaring its + // applicable model kind via the forModel annotation. + const readImage = cfg.customTools.find((t) => t.name === "read_image")!; + expect(readImage.forModel).toBe("vision"); + expect(readImage.permission).toBe("r"); + expect(Object.keys((readImage.parameters as { properties: object }).properties)).toEqual([ + "source", + ]); + const describeImage = cfg.customTools.find((t) => t.name === "describe_image")!; + expect(describeImage.forModel).toBe("text-only"); + expect(describeImage.permission).toBe("r"); + expect(Object.keys((describeImage.parameters as { properties: object }).properties)).toEqual([ + "source", + "prompt", + ]); + expect((describeImage.parameters as { required?: string[] }).required).toEqual(["source"]); + }); + + it("selectBuiltinToolsForModel picks the matching image tool per model kind", async () => { + const state = await loadOrInitAgentState(); + const all = buildToolConfig(state).customTools; + // Vision model: read_image is kept, describe_image is filtered out; unannotated tools are unaffected. + const forVision = selectBuiltinToolsForModel(all, true); + expect(forVision.some((t) => t.name === "read_image")).toBe(true); + expect(forVision.some((t) => t.name === "describe_image")).toBe(false); + expect(forVision.filter((t) => t.name === "exec_command")).toHaveLength(1); + // Text-only model: describe_image is kept. + const forText = selectBuiltinToolsForModel(all, false); + expect(forText.some((t) => t.name === "read_image")).toBe(false); + expect(forText.some((t) => t.name === "describe_image")).toBe(true); + expect(forText.filter((t) => t.name === "exec_command")).toHaveLength(1); + }); + + it("loads MCP Server config from system_config.yaml", () => { + const state = { + root: tmpRoot, + projectId: DEFAULT_PROJECT_ID, + agentId: DEFAULT_AGENT_ID, + stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + systemConfig: { + system_prompt: "x", + tools: { + builtin: [], + mcpServers: [{ name: "fs", config: { command: "mcp-fs" } }], + }, + }, + agentsMd: "y", + }; + + const cfg = buildToolConfig(state); + expect(cfg.customTools).toEqual([]); + expect(cfg.mcpServers).toEqual([{ name: "fs", config: { command: "mcp-fs" } }]); + }); + + it("falls back to default builtin tools when config omits them", () => { + const state = { + root: tmpRoot, + projectId: DEFAULT_PROJECT_ID, + agentId: DEFAULT_AGENT_ID, + stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + systemConfig: { system_prompt: "x" }, + agentsMd: "y", + }; + const cfg = buildToolConfig(state); + expect(cfg.customTools.map((t) => t.name)).toEqual([ + "exec_command", + "input_command", + "run_subagent", + "input_subagent", + "read_image", + "describe_image", + ]); + }); +}); + +describe("assembleSystemPrompt", () => { + it("renders default system prompt placeholders", async () => { + const state = await loadOrInitAgentState(); + const prompt = assembleSystemPrompt( + state, + sessionEnvironment("/tmp/penguin-ws", "session-test-1", { + agentId: DEFAULT_AGENT_ID, + projectDir: "/tmp/proj", + }), + ); + expect(prompt).toContain("AGENTS.md"); + expect(prompt).toContain("PenguinHarness"); + // File system's two file-delivery conventions: a workspace file is mentioned in the reply + // by its **workspace-relative path** in backticks (the frontend renders a message file card + // from this); scratchpad only holds intermediate artifacts, and final deliverables must + // land in the Workspace. + expect(prompt).toContain("mention its workspace-relative path in backticks"); + expect(prompt).toContain("always place final deliverables in the workspace"); + // The default template wraps AGENTS.md in a XML block. + expect(prompt).toContain(""); + expect(prompt).toContain(""); + expect(prompt.indexOf("")).toBeLessThan( + prompt.indexOf("# Environment"), + ); + expect(prompt).not.toContain(AGENTS_MD_PLACEHOLDER); + expect(prompt).not.toContain(SESSION_ID_PLACEHOLDER); + expect(prompt).not.toContain(CWD_PLACEHOLDER); + expect(prompt).not.toContain(PLATFORM_PLACEHOLDER); + expect(prompt).not.toContain(OS_VERSION_PLACEHOLDER); + expect(prompt).not.toContain(DATE_PLACEHOLDER); + }); + + it("replaces AGENTS.md and specific Session environment fields at template locations", () => { + const state = { + root: tmpRoot, + projectId: DEFAULT_PROJECT_ID, + agentId: DEFAULT_AGENT_ID, + stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + systemConfig: { + system_prompt: [ + "before", + `sid=${SESSION_ID_PLACEHOLDER}`, + `cwd=${CWD_PLACEHOLDER}`, + `aid=${AGENT_ID_PLACEHOLDER}`, + `pdir=${PROJECT_DIR_PLACEHOLDER}`, + `platform=${PLATFORM_PLACEHOLDER}`, + `os=${OS_VERSION_PLACEHOLDER}`, + `date=${DATE_PLACEHOLDER}`, + "middle", + AGENTS_MD_PLACEHOLDER, + "after", + ].join("\n"), + }, + agentsMd: "# Agent Rules\nFollow local rules.", + }; + + const prompt = assembleSystemPrompt(state, { + sessionId: "session-1", + cwd: "/tmp/ws", + agentId: "agent-x", + projectDir: "/tmp/proj", + platform: "darwin", + osVersion: "Darwin 25.0.0", + date: "2026-06-30", + }); + expect(prompt).toBe( + [ + "before", + "sid=session-1", + "cwd=/tmp/ws", + "aid=agent-x", + "pdir=/tmp/proj", + "platform=darwin", + "os=Darwin 25.0.0", + "date=2026-06-30", + "middle", + "# Agent Rules\nFollow local rules.", + "after", + ].join("\n"), + ); + }); + + it("replaces the placeholder with an empty string when AGENTS.md is blank", () => { + const state = { + root: tmpRoot, + projectId: DEFAULT_PROJECT_ID, + agentId: DEFAULT_AGENT_ID, + stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + systemConfig: { + system_prompt: ["before", AGENTS_MD_PLACEHOLDER, "after"].join("\n"), + }, + agentsMd: " \n", + }; + + const prompt = assembleSystemPrompt(state); + expect(prompt).toBe("before\n\nafter"); + }); + + it("does not append AGENTS.md or Session environment without placeholders", () => { + const state = { + root: tmpRoot, + projectId: DEFAULT_PROJECT_ID, + agentId: DEFAULT_AGENT_ID, + stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + systemConfig: { system_prompt: "base prompt" }, + agentsMd: "# Agent Rules\nShould not appear.", + }; + + const prompt = assembleSystemPrompt(state, { + sessionId: "session-1", + cwd: "/tmp/ws", + agentId: "agent-x", + projectDir: "/tmp/proj", + platform: "darwin", + osVersion: "Darwin 25.0.0", + date: "2026-06-30", + }); + expect(prompt).toBe("base prompt"); + }); + + it("renders vault key names (never values) via the {{VAULT_KEYS}} placeholder", () => { + const state = { + root: tmpRoot, + projectId: DEFAULT_PROJECT_ID, + agentId: DEFAULT_AGENT_ID, + stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + systemConfig: { + system_prompt: ["before", AGENTS_MD_PLACEHOLDER, VAULT_KEYS_PLACEHOLDER, "after"].join( + "\n", + ), + }, + agentsMd: "# Agent Rules", + }; + + const prompt = assembleSystemPrompt(state, undefined, ["KEY_A", "KEY_B"]); + // The placeholder is replaced with a list of key names (one `- KEY` per line); the vault's + // purpose statement is part of the template body, not carried by the replacement value. + expect(prompt).toBe(["before", "# Agent Rules", "- KEY_A", "- KEY_B", "after"].join("\n")); + }); + + it("replaces {{VAULT_KEYS}} with an empty string when there are no keys", () => { + const state = { + root: tmpRoot, + projectId: DEFAULT_PROJECT_ID, + agentId: DEFAULT_AGENT_ID, + stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), + systemConfig: { + system_prompt: ["before", AGENTS_MD_PLACEHOLDER, VAULT_KEYS_PLACEHOLDER, "after"].join( + "\n", + ), + }, + agentsMd: "# Agent Rules", + }; + // No keys: the placeholder is replaced with an empty string (the template body's vault + // statement is kept, though this test's template does not include one); the placeholder + // leaves no residue. + const empty = assembleSystemPrompt(state, undefined, []); + expect(empty).toBe(["before", "# Agent Rules", "", "after"].join("\n")); + expect(assembleSystemPrompt(state)).not.toContain(VAULT_KEYS_PLACEHOLDER); + }); + + it("does not auto-inject other Agent State files", async () => { + const state = await loadOrInitAgentState(); + await fs.writeFile( + path.join(memoryDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), "note.md"), + "MEMORY_SHOULD_NOT_BE_IN_PROMPT", + "utf8", + ); + await fs.writeFile( + path.join(skillsDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), "SKILL.md"), + "SKILL_SHOULD_NOT_BE_IN_PROMPT", + "utf8", + ); + + const reloaded = await loadOrInitAgentState(); + const prompt = assembleSystemPrompt(reloaded); + + expect(prompt).not.toContain("MEMORY_SHOULD_NOT_BE_IN_PROMPT"); + expect(prompt).not.toContain("SKILL_SHOULD_NOT_BE_IN_PROMPT"); + }); + + it("replaces generated Session environment field placeholders when provided", async () => { + const state = await loadOrInitAgentState(); + const env = sessionEnvironment( + "/tmp/penguin-ws", + "session-test-1", + { agentId: "agent-x", projectDir: "/tmp/proj" }, + new Date("2026-06-30T00:00:00"), + ); + const prompt = assembleSystemPrompt(state, env); + + expect(prompt).toContain("# Environment"); + expect(prompt).toContain("Session ID: session-test-1"); + expect(prompt).toContain("CWD: /tmp/penguin-ws"); + expect(prompt).toContain("Agent ID: agent-x"); + expect(prompt).toContain("Project Dir: /tmp/proj"); + expect(prompt).toContain("Platform:"); + expect(prompt).toContain("OS Version:"); + expect(prompt).toContain("Date: 2026-06-30"); + expect(prompt.indexOf("Platform:")).toBeLessThan(prompt.indexOf("OS Version:")); + expect(prompt.indexOf("OS Version:")).toBeLessThan(prompt.indexOf("Date:")); + expect(prompt.indexOf("Date:")).toBeLessThan(prompt.indexOf("CWD:")); + expect(prompt.indexOf("CWD:")).toBeLessThan(prompt.indexOf("Agent ID:")); + expect(prompt.indexOf("Agent ID:")).toBeLessThan(prompt.indexOf("Project Dir:")); + expect(prompt.indexOf("Project Dir:")).toBeLessThan(prompt.indexOf("Session ID:")); + }); +}); + +describe("project-config round trip", () => { + it("returns default config when file is absent (without writing)", async () => { + const cfg = await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID); + expect(cfg).toEqual(defaultProjectConfig()); + // loadProjectConfig must not write to disk. + expect(await exists(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID))).toBe(false); + }); + + it("persists addModel with inline credential and default, then reads back", async () => { + const saved = await addModel( + tmpRoot, + DEFAULT_PROJECT_ID, + { + provider: "custom", + model_id: "gpt-test", + context_window: 128000, + api_key: "sk-abc", + base_url: "https://example.com/v1", + }, + { setDefault: true }, + ); + // default_model is a pair reference (no string concatenation involved). + expect(saved.default_model).toEqual({ provider: "custom", model_id: "gpt-test" }); + + const loaded = await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID); + expect(loaded.default_model).toEqual({ provider: "custom", model_id: "gpt-test" }); + + // The credential is inlined in the entry; the two independent fields provider and + // model_id together form the unique key. + const entry = getModel(loaded, { provider: "custom", model_id: "gpt-test" }); + expect(entry).toEqual({ + provider: "custom", + model_id: "gpt-test", + context_window: 128000, + api_key: "sk-abc", + base_url: "https://example.com/v1", + }); + + // getModel matches the exact pair: a different provider means no match. + expect(getModel(loaded, { provider: "openai", model_id: "gpt-test" })).toBeUndefined(); + expect(getModel(loaded, { provider: "custom", model_id: "unknown-model" })).toBeUndefined(); + }); + + it("infers the provider from the builtin catalog when addModel omits it", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { model_id: "claude-sonnet-4-6" }); + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { model_id: "my-own-model" }); + const loaded = await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID); + // A catalog hit gets its provider; otherwise it falls back to custom. + expect( + getModel(loaded, { provider: "anthropic", model_id: "claude-sonnet-4-6" }), + ).toBeDefined(); + expect(getModel(loaded, { provider: "custom", model_id: "my-own-model" })).toBeDefined(); + }); + + it("upserts by the (provider, model_id) pair; same model_id under two providers co-exists", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "pa", + model_id: "m1", + context_window: 1000, + }); + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "pa", + model_id: "m1", + context_window: 2000, + }); + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "pb", + model_id: "m1", + context_window: 3000, + }); + const loaded = await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID); + const matches = loaded.models.filter((m) => m.model_id === "m1"); + expect(matches).toHaveLength(2); + expect(getModel(loaded, { provider: "pa", model_id: "m1" })?.context_window).toBe(2000); + expect(getModel(loaded, { provider: "pb", model_id: "m1" })?.context_window).toBe(3000); + }); + + it("addModel persists vision flag and upsert preserves it when not re-specified", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "ds", + vision: false, + }); + // Only supplements context_window, without vision: the original annotation is kept. + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "ds", + context_window: 64000, + }); + const dsRef = { provider: "custom", model_id: "ds" }; + let m = getModel(await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID), dsRef); + expect(m?.vision).toBe(false); + expect(m?.context_window).toBe(64000); + // Explicitly switches it back to supported. + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "ds", + vision: true, + }); + m = getModel(await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID), dsRef); + expect(m?.vision).toBe(true); + }); + + it("setVisionModel persists and validates the target", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "vis", + vision: true, + }); + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "blind", + vision: false, + }); + const visRef = { provider: "custom", model_id: "vis" }; + await setVisionModel(tmpRoot, DEFAULT_PROJECT_ID, visRef); + const loaded = await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID); + expect(loaded.vision_model).toEqual(visRef); + // A subsequent addModel save/reload round trip does not lose vision_model (loadProjectConfig + // passes it through explicitly). + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "vis", + context_window: 1000, + }); + expect((await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID)).vision_model).toEqual(visRef); + // A target that does not exist or is annotated as not supporting images: throws + // (the error includes the pair reference). + await expect( + setVisionModel(tmpRoot, DEFAULT_PROJECT_ID, { provider: "custom", model_id: "nope" }), + ).rejects.toThrow(/model_id=nope/); + await expect( + setVisionModel(tmpRoot, DEFAULT_PROJECT_ID, { provider: "custom", model_id: "blind" }), + ).rejects.toThrow(/不支持图片/); + }); + + it("upsert preserves existing context_window and inline credential when not re-specified", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "m1", + context_window: 200000, + base_url: "https://gw.example", + }); + // Only supplements an api_key, without context_window/base_url. + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "m1", + api_key: "sk-xyz", + }); + const loaded = await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID); + const m = getModel(loaded, { provider: "custom", model_id: "m1" }); + expect(m?.context_window).toBe(200000); // Not cleared + expect(m?.api_key).toBe("sk-xyz"); + expect(m?.base_url).toBe("https://gw.example"); // The original base_url is kept + }); + + it("upsert preserves display_name / created_at written by the interface layer", async () => { + // The interface layer (server) writes display_name / created_at onto an entry; the CLI-side + // addModel must not clear them when supplementing other fields (with a single config file, + // these fields now live in the same entry as the credential). + const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID); + await fs.mkdir(path.dirname(file), { recursive: true }); + await fs.writeFile( + file, + [ + "[[models]]", + 'provider = "custom"', + 'model_id = "m-keep"', + 'display_name = "My Model"', + 'api_key = "sk-old"', + 'created_at = "2026-07-01T00:00:00Z"', + ].join("\n"), + "utf8", + ); + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "m-keep", + api_key: "sk-new", + }); + const m = getModel(await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID), { + provider: "custom", + model_id: "m-keep", + }); + expect(m?.api_key).toBe("sk-new"); + expect(m?.display_name).toBe("My Model"); + expect(m?.created_at).toBe("2026-07-01T00:00:00Z"); + }); + + it("default config carries the anthropic claude-sonnet-4-6 pricing (three buckets)", () => { + const entry = getModel(defaultProjectConfig(), { + provider: "anthropic", + model_id: "claude-sonnet-4-6", + }); + expect(entry?.context_window).toBe(1000000); + expect(entry?.pricing).toEqual({ + unit: "usd_per_mtok", + cache_read: 0.3, + cache_write: 3.75, + output: 15, + }); + // A preset model that supports vision does not persist a vision field (default = supported). + expect(entry?.vision).toBeUndefined(); + }); + + it("default config presets the full model catalog (default = deepseek deepseek-v4-pro)", () => { + const cfg = defaultProjectConfig(); + expect(cfg.default_model).toEqual({ provider: "deepseek", model_id: "deepseek-v4-pro" }); + // The catalog is presented in full: provider and model_id are separate columns, model_id + // being the plain upstream id (vision is only persisted as false for models that don't + // support images). + expect(cfg.models.map((m) => [m.provider, m.model_id])).toEqual( + MODEL_CATALOG.map((m) => [m.provider, m.modelId]), + ); + for (const entry of cfg.models) { + const cat = MODEL_CATALOG.find( + (c) => c.provider === entry.provider && c.modelId === entry.model_id, + )!; + expect(entry.vision).toBe(cat.supportsVision ? undefined : false); + expect(entry.pricing?.unit).toBe("usd_per_mtok"); + // A model that auto-routes leaves client_type unset; a gateway model (OpenRouter) + // explicitly sets it to openai. + expect(entry.client_type).toBe(cat.clientType); + // A gateway model has its base URL preset inline (no key included); other models have + // no credential. + expect(entry.base_url).toBe(cat.baseUrl); + expect(entry.api_key).toBeUndefined(); + } + expect(getModel(cfg, { provider: "openrouter", model_id: "xiaomi/mimo-v2.5" })?.base_url).toBe( + "https://openrouter.ai/api/v1", + ); + expect( + getModel(cfg, { provider: "deepseek", model_id: "deepseek-v4-pro" })?.base_url, + ).toBeUndefined(); + }); + + it("persists pricing and field-merges buckets on upsert", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "p1", + pricing: { cache_read: 0.3, cache_write: 3.75, output: 15 }, + }); + // Only output is updated, the other two buckets are kept. + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "p1", + pricing: { output: 20 }, + }); + const loaded = await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID); + expect(getModel(loaded, { provider: "custom", model_id: "p1" })?.pricing).toEqual({ + unit: "usd_per_mtok", + cache_read: 0.3, + cache_write: 3.75, + output: 20, + }); + }); + + it("setDefaultModel updates and persists a pair reference", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "m2", + context_window: 4096, + }); + const m2Ref = { provider: "custom", model_id: "m2" }; + const updated = await setDefaultModel(tmpRoot, DEFAULT_PROJECT_ID, m2Ref); + expect(updated.default_model).toEqual(m2Ref); + const loaded = await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID); + expect(loaded.default_model).toEqual(m2Ref); + // A target not in models: throws (the same validation as setVisionModel, with the error + // including the pair reference and a model-list hint), and the original default model is + // unaffected. A mismatched provider likewise fails (exact pair match, no fuzzy resolution). + await expect( + setDefaultModel(tmpRoot, DEFAULT_PROJECT_ID, { provider: "custom", model_id: "nope" }), + ).rejects.toThrow(/\(provider=custom, model_id=nope\).*model list/); + await expect( + setDefaultModel(tmpRoot, DEFAULT_PROJECT_ID, { provider: "openai", model_id: "m2" }), + ).rejects.toThrow(/不在 models 中/); + expect((await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID)).default_model).toEqual(m2Ref); + }); + + it("loadProjectConfig tolerates an empty config file (returns defaults, no throw)", async () => { + const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID); + await fs.mkdir(path.dirname(file), { recursive: true }); + await fs.writeFile(file, "", "utf8"); // Empty file -> parseToml may return null + const loaded = await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID); + expect(loaded.models).toEqual([]); + }); + + it("rejects old-format config files with a clear error (no migration)", async () => { + const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID); + await fs.mkdir(path.dirname(file), { recursive: true }); + // Old format 1: default_model is a concatenated storage id string. + await fs.writeFile(file, 'default_model = "deepseek/deepseek-v4-pro"\n', "utf8"); + await expect(loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID)).rejects.toThrow(/旧版本|成对引用/); + // Old format 2: a model entry missing provider (from the era of composite model_id + + // request_model_id). + await fs.writeFile( + file, + [ + "[[models]]", + 'model_id = "anthropic/claude-sonnet-4-6"', + 'request_model_id = "claude-sonnet-4-6"', + ].join("\n"), + "utf8", + ); + await expect(loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID)).rejects.toThrow(/旧版本|独立字段/); + }); +}); + +describe("resolveModelRef(唯一的「省略 provider」解析入口)", () => { + const cfg: ProjectConfig = { + models: [ + { provider: "deepseek", model_id: "deepseek-v4-pro" }, + { provider: "siliconflow", model_id: "shared-id" }, + { provider: "openrouter", model_id: "shared-id" }, + ], + }; + + it("resolves a globally unique model_id without provider (exact match only)", () => { + expect(resolveModelRef(cfg, "deepseek-v4-pro")).toEqual({ + provider: "deepseek", + model_id: "deepseek-v4-pro", + }); + // Exact-match lookup, no fuzzy/prefix matching. + expect(() => resolveModelRef(cfg, "deepseek-v4")).toThrow(/不在 Project 配置中/); + }); + + it("validates the exact pair when provider is given", () => { + expect(resolveModelRef(cfg, "shared-id", "siliconflow")).toEqual({ + provider: "siliconflow", + model_id: "shared-id", + }); + // A wrong provider grouping likewise fails: the error includes the pair reference. + expect(() => resolveModelRef(cfg, "shared-id", "openai")).toThrow( + /\(provider=openai, model_id=shared-id\)/, + ); + }); + + it("errors clearly on zero matches", () => { + expect(() => resolveModelRef(cfg, "no-such-model")).toThrow( + /不在 Project 配置中.*no-such-model/, + ); + }); + + it("errors on ambiguity, listing the candidate pair references", () => { + expect(() => resolveModelRef(cfg, "shared-id")).toThrow(/歧义/); + expect(() => resolveModelRef(cfg, "shared-id")).toThrow( + /\(provider=siliconflow, model_id=shared-id\).*\(provider=openrouter, model_id=shared-id\)/, + ); + }); +}); + +describe("单一隐藏配置文件(.project_config.toml,credential 内联)", () => { + it("addModel writes one hidden file with 0600 permission; api_key lives inline", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "m-split", + api_key: "sk-split-1", + }); + const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID); + // The sole config file is hidden (not shown by ls by default) and has 0600 permission + // (owner read/write only). + expect(path.basename(file)).toBe(".project_config.toml"); + expect((await fs.stat(file)).mode & 0o777).toBe(0o600); + expect(await fs.readFile(file, "utf8")).toContain("sk-split-1"); + // The old two-file layout is no longer produced. + expect(await exists(path.join(tmpRoot, DEFAULT_PROJECT_ID, "project_config.toml"))).toBe(false); + expect(await exists(path.join(tmpRoot, DEFAULT_PROJECT_ID, ".credentials.toml"))).toBe(false); + }); + + it("chmod converges an existing file back to 0600 on save", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "m-perm", + api_key: "sk-1", + }); + const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID); + await fs.chmod(file, 0o644); + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "m-perm", + api_key: "sk-2", + }); + expect((await fs.stat(file)).mode & 0o777).toBe(0o600); + }); + + it("writes provider and model_id as separate fields; refs are TOML inline tables", async () => { + await addModel( + tmpRoot, + DEFAULT_PROJECT_ID, + { provider: "openrouter", model_id: "xiaomi/mimo-v2.5" }, + { setDefault: true }, + ); + await setVisionModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "anthropic", + model_id: "claude-sonnet-4-6", + }); + const raw = await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"); + // A reference pair is persisted as an inline table; entries have + // provider / model_id as separate columns, with no concatenated storage id or + // request_model_id appearing anywhere. + expect(raw).toContain( + 'default_model = { provider = "openrouter", model_id = "xiaomi/mimo-v2.5" }', + ); + expect(raw).toContain( + 'vision_model = { provider = "anthropic", model_id = "claude-sonnet-4-6" }', + ); + expect(raw).toContain('provider = "openrouter"'); + expect(raw).toContain('model_id = "xiaomi/mimo-v2.5"'); + expect(raw).not.toContain("request_model_id"); + expect(raw).not.toContain('"openrouter/xiaomi/mimo-v2.5"'); + }); +}); + +describe("agent vault (agent_state/.vault.toml)", () => { + it("set/remove roundtrip persists to the agent's .vault.toml", async () => { + await setVaultEntry(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "MY_API_KEY", "sk-secret-1"); + await setVaultEntry(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "OTHER_KEY", "v2"); + // A same-named key overwrites, producing no duplicate. + await setVaultEntry(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "MY_API_KEY", "sk-secret-2"); + let vault = await loadAgentVault(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID); + expect(vault).toEqual({ MY_API_KEY: "sk-secret-2", OTHER_KEY: "v2" }); + // Persisted in plaintext to this Agent's agent_state/.vault.toml (an accepted tradeoff: + // masking happens at the interface layer) -- a hidden file (not shown by ls by default) + // with 0600 permission (owner read/write only). + const file = agentVaultPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID); + expect(path.basename(file)).toBe(".vault.toml"); + expect((await fs.stat(file)).mode & 0o777).toBe(0o600); + const raw = await fs.readFile(file, "utf8"); + expect(raw).toContain("sk-secret-2"); + // The Project config no longer carries the vault. + expect(JSON.stringify(await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID))).not.toContain( + "sk-secret-2", + ); + + vault = await removeVaultEntry(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "MY_API_KEY"); + expect(vault).toEqual({ OTHER_KEY: "v2" }); + // Once emptied, the whole .vault.toml is removed; removing a non-existent key is + // idempotent and does not throw. + vault = await removeVaultEntry(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "OTHER_KEY"); + expect(vault).toEqual({}); + await expect(fs.access(file)).rejects.toThrow(); + vault = await removeVaultEntry(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "GHOST"); + expect(vault).toEqual({}); + }); + + it("keeps vaults independent between agents", async () => { + await setVaultEntry(tmpRoot, DEFAULT_PROJECT_ID, "agent-a", "KEY_A", "va"); + await setVaultEntry(tmpRoot, DEFAULT_PROJECT_ID, "agent-b", "KEY_B", "vb"); + expect(await loadAgentVault(tmpRoot, DEFAULT_PROJECT_ID, "agent-a")).toEqual({ KEY_A: "va" }); + expect(await loadAgentVault(tmpRoot, DEFAULT_PROJECT_ID, "agent-b")).toEqual({ KEY_B: "vb" }); + }); + + it("rejects invalid keys and keeps shell-safe names only", async () => { + const set = (key: string) => + setVaultEntry(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, key, "v"); + await expect(set("1BAD")).rejects.toThrow(/vault key/); + await expect(set("BAD-DASH")).rejects.toThrow(); + await expect(set("BAD KEY")).rejects.toThrow(); + await expect(set("")).rejects.toThrow(); + // Starting with an underscore is valid (shell environment variable naming rule). + await set("_OK_1"); + expect(isValidVaultKey("_OK_1")).toBe(true); + expect(isValidVaultKey("9NOPE")).toBe(false); + // A value that is too long (>8192) is rejected: since it gets injected into the child + // process environment, an oversized value would make exec spawn fail (E2BIG). + await expect( + setVaultEntry(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "OK_BIG", "x".repeat(8193)), + ).rejects.toThrow(/too long/); + }); + + it("ignores non-string values and invalid key names from a hand-edited TOML; missing file is an empty vault", async () => { + const file = agentVaultPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID); + await fs.mkdir(path.dirname(file), { recursive: true }); + // Non-string values and invalid key names (starting with a dash / digit) are always + // ignored -- the same rule as the write side (review gemini #1: if an invalid key were + // loaded, it would get injected into the Prompt/env, and after a GET brought it out, a PUT + // of the whole table back would 400, bricking the vault page). + await fs.writeFile(file, 'GOOD = "ok"\nBAD = 123\n"BAD-DASH" = "x"\n"9NUM" = "y"\n', "utf8"); + expect(await loadAgentVault(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID)).toEqual({ + GOOD: "ok", + }); + expect(await loadAgentVault(tmpRoot, DEFAULT_PROJECT_ID, "no-such-agent")).toEqual({}); + }); +}); + +describe("defensive config parsing", () => { + it("throws a clear error when system_config.yaml is empty or corrupt", async () => { + // First initialize normally, then empty out system_config.yaml; reloading should throw a + // clear error rather than producing an undefined-laden message. + await loadOrInitAgentState(); + const cfgPath = systemConfigPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID); + await fs.writeFile(cfgPath, "", "utf8"); + await expect(loadOrInitAgentState()).rejects.toThrow(/system_prompt|非法|损坏/); + + await fs.writeFile(cfgPath, "just a string, not a mapping", "utf8"); + await expect(loadOrInitAgentState()).rejects.toThrow(/system_prompt|非法|损坏/); + }); +}); diff --git a/packages/core/test/subagent.test.ts b/packages/core/test/subagent.test.ts new file mode 100644 index 0000000..c89f486 --- /dev/null +++ b/packages/core/test/subagent.test.ts @@ -0,0 +1,515 @@ +/** + * Behavior tests for run_subagent / input_subagent: foreground delegation, backgrounding, + * polling, resuming with an appended Prompt, the approval queue, and lifecycle finalization. + */ +import { afterEach, describe, expect, it } from "vitest"; +import { createSubagentTool } from "../src/environment/tools/run-subagent.js"; +import { createInputSubagentTool } from "../src/environment/tools/input-subagent.js"; +import { + ManagedSubagentSession, + SubagentSessionManager, +} from "../src/environment/tools/subagent/index.js"; +import { abortEvent, partialText, toolCall, withOrigin } from "../src/omnimessage/index.js"; +import { collectWindow } from "../src/environment/tools/subagent/collect.js"; +import type { MessageOrigin, OmniMessage } from "../src/omnimessage/index.js"; +import type { + ApproveFn, + EnvironmentServices, + SubagentHandle, + SubagentRunner, + ToolDefinitionConfig, +} from "../src/interfaces.js"; +import type { ToolExecutionContext, ToolResult } from "../src/environment/tools/types.js"; + +const DEF: ToolDefinitionConfig = { + name: "run_subagent", + description: "delegate a subtask", + permission: "rw", +}; + +const INPUT_DEF: ToolDefinitionConfig = { + name: "input_subagent", + description: "drive a background subagent", + permission: "rw", +}; + +const CTX: ToolExecutionContext = { + workspaceDir: "/tmp/ws", + toolCallId: "call_1", +}; + +/** The origin tag the simulated runner stamps on (contract: every message handle.run yields + * already carries the child Session id). */ +const HOP: MessageOrigin = "session-child-12ab34cd"; + +interface LoosePayload { + type?: string; + event_type?: string; + output?: string; + stop_reason?: string; + tool_call_id?: string; +} +const pl = (m: OmniMessage): LoosePayload => m.payload as LoosePayload; + +type RunInput = { prompt: string; signal?: AbortSignal; approve?: ApproveFn }; + +/** Builds a SubagentRunner from a run implementation (spawn arguments observed via a spy). */ +function runnerOf( + run: (input: RunInput) => AsyncGenerator, + spawnSpy?: (input: { agentId?: string; modelId?: string }) => void, +): SubagentRunner { + return { + async spawn(input) { + spawnSpy?.(input); + const handle: SubagentHandle = { sessionId: HOP, run, dispose() {} }; + return handle; + }, + }; +} + +/** A promise that resolves when the signal aborts (never resolves if there is no signal). */ +function aborted(signal?: AbortSignal): Promise { + return new Promise((resolve) => { + if (!signal) return; + if (signal.aborted) return resolve(); + signal.addEventListener("abort", () => resolve(), { once: true }); + }); +} + +/** Polls until a condition holds (test helper). */ +async function until(cond: () => boolean, ms = 3000): Promise { + const start = Date.now(); + while (!cond()) { + if (Date.now() - start > ms) throw new Error("condition not met in time"); + await new Promise((r) => setTimeout(r, 10)); + } +} + +/** Collects yielded messages and captures the generator's return value (the tool reports its + * finish reason via the return value). */ +async function collectWithReturn( + gen: AsyncGenerator, +): Promise<{ out: OmniMessage[]; result: ToolResult | void }> { + const out: OmniMessage[] = []; + for (;;) { + const res = await gen.next(); + if (res.done) return { out, result: res.value }; + out.push(res.value); + } +} + +/** Concatenates the tool's own (origin-free) output deltas. */ +const ownDeltas = (out: OmniMessage[]): string => + out + .filter( + (m) => + !m.origin?.length && + pl(m).type === "partial_tool_call_output" && + pl(m).event_type === "delta", + ) + .map((m) => pl(m).output ?? "") + .join(""); + +/** Extracts the subagent_id from run_subagent's finishing note. */ +function extractSubagentId(result: ToolResult | void): string { + const m = (result?.note ?? "").match(/subagent_id (subagent-[0-9a-f]+)/); + expect(m, `expected a subagent_id in: ${JSON.stringify(result?.note)}`).toBeTruthy(); + return m![1]!; +} + +const managers: SubagentSessionManager[] = []; +function makeServices(runner?: SubagentRunner): { + services: EnvironmentServices; + manager: SubagentSessionManager; +} { + const manager = new SubagentSessionManager(); + managers.push(manager); + return { + services: { ...(runner ? { subagentRunner: runner } : {}), subagentSessions: manager }, + manager, + }; +} + +afterEach(() => { + for (const m of managers.splice(0)) m.dispose(); +}); + +describe("run_subagent tool (foreground)", () => { + it("forwards stamped child messages and mirrors child text as its own output deltas", async () => { + const seen: Array<{ prompt?: string; agentId?: string; modelId?: string }> = []; + const runner = runnerOf( + async function* (input) { + seen[0] = { ...seen[0], prompt: input.prompt }; + yield withOrigin(partialText("delta", "Hello "), HOP); + yield withOrigin(partialText("delta", input.prompt), HOP); + }, + (input) => { + seen[0] = { ...input }; + }, + ); + const { services } = makeServices(runner); + const tool = createSubagentTool(DEF, services); + const { out, result } = await collectWithReturn( + tool.execute({ prompt: "world", agent_id: "researcher", model_id: "m1" }, CTX), + ); + + // Child session messages pass through verbatim (with origin). + const forwarded = out.filter((m) => m.origin?.length); + expect(forwarded).toHaveLength(2); + expect(forwarded[0]!.origin![0]).toEqual(HOP); + // The child's text deltas are mirrored as this tool's own output (Environment derives the + // complete tool_call_output from this). + expect(ownDeltas(out)).toBe("Hello world"); + expect(result?.stopReason).toBe("completed"); + // The model is free to choose the agent and model (spawn arguments); the prompt is + // handed to run. + expect(seen[0]).toEqual({ prompt: "world", agentId: "researcher", modelId: "m1" }); + }); + + it("does not mirror deeper-nested (origin.length > 1) text into its own output", async () => { + const grandHop: MessageOrigin = "sess_grandchild"; + const runner = runnerOf(async function* () { + // Grandchild-level text (two hops): only forwarded, not counted as the child Agent's reply. + yield withOrigin(withOrigin(partialText("delta", "deep"), grandHop), HOP); + yield withOrigin(partialText("delta", "answer"), HOP); + }); + const { services } = makeServices(runner); + const tool = createSubagentTool(DEF, services); + const { out } = await collectWithReturn(tool.execute({ prompt: "x" }, CTX)); + expect(ownDeltas(out)).toBe("answer"); + // Grandchild-level messages are still forwarded (origin two hops). + expect(out.some((m) => (m.origin?.length ?? 0) === 2)).toBe(true); + }); + + it("fails gracefully when no runner is injected", async () => { + const { services } = makeServices(); + const tool = createSubagentTool(DEF, services); + const { out, result } = await collectWithReturn(tool.execute({ prompt: "x" }, CTX)); + expect(result?.stopReason).toBe("failed"); + expect(ownDeltas(out)).toContain("no subagent runner"); + }); + + it("fails when the required prompt is missing", async () => { + const runner = runnerOf( + // eslint-disable-next-line require-yield + async function* () { + /* never invoked */ + }, + ); + const { services } = makeServices(runner); + const tool = createSubagentTool(DEF, services); + const { out, result } = await collectWithReturn(tool.execute({}, CTX)); + expect(result?.stopReason).toBe("failed"); + expect(ownDeltas(out)).toContain("prompt"); + }); + + it("notes when the subagent produces no text", async () => { + const runner = runnerOf( + // eslint-disable-next-line require-yield + async function* () { + /* yields no assistant text */ + }, + ); + const { services } = makeServices(runner); + const tool = createSubagentTool(DEF, services); + const { result } = await collectWithReturn(tool.execute({ prompt: "x" }, CTX)); + expect(result?.stopReason).toBe("completed"); + expect(result?.note).toContain("without a text answer"); + }); + + it("reports a failed delegation when the child session aborts", async () => { + const runner = runnerOf(async function* () { + yield withOrigin(partialText("delta", "partial"), HOP); + yield withOrigin(abortEvent("llm error"), HOP); + }); + const { services } = makeServices(runner); + const tool = createSubagentTool(DEF, services); + const { result } = await collectWithReturn(tool.execute({ prompt: "x" }, CTX)); + expect(result?.stopReason).toBe("failed"); + expect(result?.note).toContain("subagent aborted: llm error"); + }); + + it("surfaces child approval requests through the parent approve callback", async () => { + const askedFor: string[] = []; + const approve: ApproveFn = async (tc) => { + askedFor.push(tc.payload.name); + expect(tc.origin?.length).toBe(1); // The approval request carries origin, so the approval UI can identify its source + return "allow"; + }; + const runner = runnerOf(async function* ({ approve: childApprove }) { + const decision = childApprove + ? await childApprove( + withOrigin(toolCall({ name: "exec_command", arguments: "{}", toolCallId: "t1" }), HOP), + ) + : "deny"; + yield withOrigin(partialText("delta", `decision:${decision}`), HOP); + }); + const { services } = makeServices(runner); + const tool = createSubagentTool(DEF, services); + const { out, result } = await collectWithReturn( + tool.execute({ prompt: "x" }, { ...CTX, approve }), + ); + expect(result?.stopReason).toBe("completed"); + expect(ownDeltas(out)).toContain("decision:allow"); + expect(askedFor).toEqual(["exec_command"]); + }); + + it("kills the child and reports aborted when interrupted during the start window", async () => { + let sawAbort = false; + const runner = runnerOf(async function* ({ signal }) { + yield withOrigin(partialText("delta", "working"), HOP); + await aborted(signal); + sawAbort = true; + }); + const { services } = makeServices(runner); + const tool = createSubagentTool(DEF, services); + const controller = new AbortController(); + setTimeout(() => controller.abort(), 100); + const { result } = await collectWithReturn( + tool.execute({ prompt: "x", yield_time_ms: 10_000 }, { ...CTX, signal: controller.signal }), + ); + expect(result?.stopReason).toBe("aborted"); + await until(() => sawAbort); + }); +}); + +describe("run_subagent backgrounding + input_subagent", () => { + /** A child Agent whose first-turn task is stuck on a gate: after backgrounding, the test + * controls when it finishes. */ + function gatedChild(): { + run: (input: RunInput) => AsyncGenerator; + release: () => void; + prompts: string[]; + } { + let release!: () => void; + const gate = new Promise((r) => (release = r)); + const prompts: string[] = []; + const run = async function* ({ prompt, signal }: RunInput): AsyncGenerator { + prompts.push(prompt); + if (prompts.length === 1) { + yield withOrigin(partialText("delta", `start:${prompt} `), HOP); + await Promise.race([gate, aborted(signal)]); + if (signal?.aborted) return; + yield withOrigin(partialText("delta", `end:${prompt}`), HOP); + return; + } + yield withOrigin(partialText("delta", `ran:${prompt}`), HOP); + }; + return { run, release, prompts }; + } + + it("yields a subagent_id when the subagent is still working past yield_time_ms", async () => { + const child = gatedChild(); + const { services, manager } = makeServices(runnerOf(child.run)); + const tool = createSubagentTool(DEF, services); + const { out, result } = await collectWithReturn( + tool.execute({ prompt: "task", yield_time_ms: 250 }, CTX), + ); + expect(result?.stopReason).toBe("completed"); + expect(ownDeltas(out)).toBe("start:task "); + const id = extractSubagentId(result); + // subagent_id is derived from the tail of the child Session id: it can be correlated with + // the message origin / frontend nesting tag. + expect(id).toBe(`subagent-${HOP.slice(-8)}`); + expect(manager.get(id)).toBeDefined(); + child.release(); + }); + + it("polls a background subagent and reports the final status when it finishes", async () => { + const child = gatedChild(); + const { services } = makeServices(runnerOf(child.run)); + const runTool = createSubagentTool(DEF, services); + const { result: started } = await collectWithReturn( + runTool.execute({ prompt: "task", yield_time_ms: 250 }, CTX), + ); + const id = extractSubagentId(started); + + child.release(); + const writeTool = createInputSubagentTool(INPUT_DEF, services); + const { out, result } = await collectWithReturn( + writeTool.execute({ subagent_id: id, yield_time_ms: 3000 }, CTX), + ); + // Output buffered while backgrounded is delivered on the tail via polling; after a turn + // ends, the session is kept (resumable). + expect(ownDeltas(out)).toContain("end:task"); + expect(result?.stopReason).toBe("completed"); + expect(result?.note).toContain(`subagent idle with subagent_id ${id}`); + }); + + it("continues the same subagent session with a follow-up prompt", async () => { + const child = gatedChild(); + const { services } = makeServices(runnerOf(child.run)); + const runTool = createSubagentTool(DEF, services); + const { result: started } = await collectWithReturn( + runTool.execute({ prompt: "one", yield_time_ms: 250 }, CTX), + ); + const id = extractSubagentId(started); + child.release(); + const writeTool = createInputSubagentTool(INPUT_DEF, services); + await collectWithReturn(writeTool.execute({ subagent_id: id, yield_time_ms: 3000 }, CTX)); + + // Appends a Prompt: resumes for a second turn on the same child Session. + const { out, result } = await collectWithReturn( + writeTool.execute({ subagent_id: id, prompt: "two", yield_time_ms: 3000 }, CTX), + ); + expect(ownDeltas(out)).toContain("ran:two"); + expect(result?.stopReason).toBe("completed"); + expect(child.prompts).toEqual(["one", "two"]); + }); + + it("rejects a follow-up prompt while the subagent is still running", async () => { + const child = gatedChild(); + const { services } = makeServices(runnerOf(child.run)); + const runTool = createSubagentTool(DEF, services); + const { result: started } = await collectWithReturn( + runTool.execute({ prompt: "task", yield_time_ms: 250 }, CTX), + ); + const id = extractSubagentId(started); + + const writeTool = createInputSubagentTool(INPUT_DEF, services); + const { out, result } = await collectWithReturn( + writeTool.execute({ subagent_id: id, prompt: "more", yield_time_ms: 250 }, CTX), + ); + expect(result?.stopReason).toBe("failed"); + expect(ownDeltas(out)).toContain("still running"); + child.release(); + }); + + it("reports an unknown subagent_id without throwing", async () => { + const { services } = makeServices(); + const writeTool = createInputSubagentTool(INPUT_DEF, services); + const { out, result } = await collectWithReturn( + writeTool.execute({ subagent_id: "subagent-deadbeef" }, CTX), + ); + expect(result?.stopReason).toBe("failed"); + expect(ownDeltas(out)).toContain("unknown subagent_id subagent-deadbeef"); + }); + + it("queues child approvals while backgrounded and surfaces them on the next poll", async () => { + const runner = runnerOf(async function* ({ approve }: RunInput) { + yield withOrigin(partialText("delta", "working "), HOP); + const decision = approve + ? await approve( + withOrigin(toolCall({ name: "exec_command", arguments: "{}", toolCallId: "t1" }), HOP), + ) + : "deny"; + yield withOrigin(partialText("delta", `approved:${decision}`), HOP); + }); + const { services } = makeServices(runner); + // The start call has no approve: once the window ends and it backgrounds, the child + // session's approval request queues up waiting. + const runTool = createSubagentTool(DEF, services); + const { result: started } = await collectWithReturn( + runTool.execute({ prompt: "task", yield_time_ms: 250 }, CTX), + ); + const id = extractSubagentId(started); + expect(started?.note).toContain("waiting for approval of 1 tool call(s)"); + + // Polling hooks up the approval outlet: the queued request is put to ctx.approve, and the + // decision is sent back to the child session. + const approve: ApproveFn = async () => "allow"; + const writeTool = createInputSubagentTool(INPUT_DEF, services); + const { out, result } = await collectWithReturn( + writeTool.execute({ subagent_id: id, yield_time_ms: 3000 }, { ...CTX, approve }), + ); + expect(ownDeltas(out)).toContain("approved:allow"); + expect(result?.stopReason).toBe("completed"); + }); + + it("refuses to spawn beyond the background subagent capacity", async () => { + const { services, manager } = makeServices( + runnerOf(async function* ({ signal }) { + yield withOrigin(partialText("delta", "x"), HOP); + await aborted(signal); + }), + ); + // Fills the concurrency limit: 8 running background sessions (running ones cannot be evicted). + for (let i = 0; i < 8; i += 1) { + const session = new ManagedSubagentSession({ + sessionId: `session-occupy-0000000${i}`, + // eslint-disable-next-line require-yield + run: async function* ({ signal }: RunInput): AsyncGenerator { + await aborted(signal); + }, + dispose() {}, + }); + session.startRun("occupy"); + manager.register(session); + } + const tool = createSubagentTool(DEF, services); + const { out, result } = await collectWithReturn(tool.execute({ prompt: "x" }, CTX)); + expect(result?.stopReason).toBe("failed"); + expect(ownDeltas(out)).toContain("too many background subagents"); + }); + + it("aborts background subagents and denies pending approvals on dispose", async () => { + let sawAbort = false; + let decision: string | null = null; + const runner = runnerOf(async function* ({ approve, signal }: RunInput) { + yield withOrigin(partialText("delta", "working"), HOP); + if (approve) { + decision = await approve( + withOrigin(toolCall({ name: "exec_command", arguments: "{}", toolCallId: "t1" }), HOP), + ); + } + await aborted(signal); + sawAbort = true; + }); + const { services, manager } = makeServices(runner); + const tool = createSubagentTool(DEF, services); + const { result } = await collectWithReturn( + tool.execute({ prompt: "task", yield_time_ms: 250 }, CTX), + ); + extractSubagentId(result); + + manager.dispose(); + await until(() => sawAbort); + expect(decision).toBe("deny"); + }); + + it("delivers output arriving while the consumer is suspended without waiting out the window", async () => { + // Wake-race regression: when output arrives while suspended at `yield`, its wakeup happens + // before the next wait begins (so it would be missed). collectWindow must re-check the + // buffer right before sleeping, otherwise this batch of output would not be delivered + // until the window ends (here, 5s). + let emitSecond: (() => void) | null = null; + const session = new ManagedSubagentSession({ + sessionId: HOP, + run: async function* ({ signal }: RunInput): AsyncGenerator { + yield withOrigin(partialText("delta", "first"), HOP); + await new Promise((resolve) => { + emitSecond = resolve; + }); + yield withOrigin(partialText("delta", "second"), HOP); + await aborted(signal); + }, + dispose() {}, + }); + try { + session.startRun("go"); + const gen = collectWindow(session, { yieldMs: 5000, toolCallId: "call_race" }); + const first = await gen.next(); // First: the forwarded "first" child session message + expect(first.done).toBe(false); + // The generator is still suspended at the yield above: releasing "second" now means both + // buffering and the wakeup have already happened. + await until(() => emitSecond !== null); + emitSecond!(); + await until(() => session.hasPending); + const startedAt = Date.now(); + let out = ""; + for (;;) { + const res = await gen.next(); + expect(res.done).toBe(false); + const p = pl(res.value as OmniMessage); + if (p.type === "partial_text" || p.type === "partial_tool_call_output") { + out += (res.value.payload as { text?: string; output?: string }).text ?? p.output ?? ""; + } + if (out.includes("second")) break; + } + expect(Date.now() - startedAt).toBeLessThan(1500); + await gen.return(undefined); + } finally { + session.kill(); + } + }); +}); diff --git a/packages/core/test/trace.test.ts b/packages/core/test/trace.test.ts new file mode 100644 index 0000000..b690b1e --- /dev/null +++ b/packages/core/test/trace.test.ts @@ -0,0 +1,160 @@ +import { mkdtemp, rm, stat } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join, sep } from "node:path"; + +import { afterEach, beforeEach, describe, expect, it } from "vitest"; + +import { + assistantText, + partialText, + sessionMeta, + subagentEvent, + tokenUsage, + emptyTokenCounts, + withOrigin, +} from "../src/omnimessage/index.js"; +import { Writer, readTrace } from "../src/trace/index.js"; + +const SESSION_ID = "sess_abc"; + +function meta() { + return sessionMeta({ + session_id: SESSION_ID, + provider: "custom", + model_id: "test-model", + model_context_window: 200000, + system_prompt: "test system prompt", + tools: [{ name: "exec_command", description: "test tool" }], + thinking_level: "medium", + agent_state: "/tmp/agent_state", + workspace: "/tmp/workspace", + }); +} + +describe("Writer", () => { + let tracesDir: string; + + beforeEach(async () => { + tracesDir = await mkdtemp(join(tmpdir(), "penguin-trace-")); + }); + + afterEach(async () => { + await rm(tracesDir, { recursive: true, force: true }); + }); + + it("writes only recordable messages and skips partial_*, in order", async () => { + // Injects a fixed date, and asserts the directory name. + const writer = new Writer({ + tracesDir, + sessionId: SESSION_ID, + date: new Date(2026, 0, 9), // Local 2026-01-09 (note the zero padding) + }); + + await writer.writeAll([ + meta(), + assistantText("hi"), + partialText("delta", "x"), // Should be skipped + tokenUsage(emptyTokenCounts(), emptyTokenCounts()), + ]); + + const rows = await readTrace(writer.currentPath()); + + // Exactly 3 rows (partial_text is skipped). + expect(rows).toHaveLength(3); + + // Every row can be JSON.parse'd (readTrace already parses it) and is in the correct order. + expect(rows[0]!.type).toBe("session_meta"); + expect(rows[1]!.type).toBe("model_msg"); + expect((rows[1]!.payload as { type: string }).type).toBe("text"); + expect(rows[2]!.type).toBe("event_msg"); + expect((rows[2]!.payload as { type: string }).type).toBe("token_usage"); + + // Contains no partial_* at all. + const innerTypes = rows.map((m) => (m.payload as { type?: string }).type); + expect(innerTypes.some((t) => t?.startsWith("partial_"))).toBe(false); + }); + + it("skips all nested-session messages; the subagent pointer event is recordable", async () => { + const writer = new Writer({ + tracesDir, + sessionId: SESSION_ID, + date: new Date(2026, 0, 9), + }); + const childMeta = sessionMeta({ + session_id: "sess_child", + provider: "custom", + model_id: "test-model", + model_context_window: 200000, + system_prompt: "child prompt", + tools: [], + thinking_level: "medium", + agent_state: "/tmp/child_agent/agent_state", + workspace: "/tmp/workspace", + }); + await writer.writeAll([ + meta(), + // The derived pointer is written by context_engine when the child session_meta arrives + // (recording only the child Session id). + subagentEvent("sess_child"), + withOrigin(childMeta, "sess_child"), + withOrigin(assistantText("from child"), "sess_child"), + withOrigin(withOrigin(childMeta, "sess_grandchild"), "sess_child"), + assistantText("from parent"), + ]); + const rows = await readTrace(writer.currentPath()); + // meta + the subagent pointer event + the parent's text; origin-tagged child session + // messages (including session_meta) are never written. + expect(rows).toHaveLength(3); + expect(rows.some((m) => (m.payload as { text?: string }).text === "from child")).toBe(false); + expect(rows.some((m) => m.origin !== undefined)).toBe(false); + const pointer = rows[1]!; + expect(pointer.type).toBe("event_msg"); + expect(pointer.payload).toMatchObject({ type: "subagent", session_id: "sess_child" }); + }); + + it("uses padded date subdir and _001.jsonl path", async () => { + const writer = new Writer({ + tracesDir, + sessionId: SESSION_ID, + date: new Date(2026, 0, 9), + }); + await writer.write(meta()); + + const path = writer.currentPath(); + // The path matches _001.jsonl and sits under a / subdirectory. + const parts = path.split(sep); + const fileName = parts[parts.length - 1]!; + const dateSubdir = parts[parts.length - 2]!; + expect(fileName).toBe(`${SESSION_ID}_001.jsonl`); + expect(dateSubdir).toBe("2026-01-09"); + + // The file actually exists. + const info = await stat(path); + expect(info.isFile()).toBe(true); + }); + + it("rotate() switches to _002.jsonl and leaves the old file append-only", async () => { + const writer = new Writer({ + tracesDir, + sessionId: SESSION_ID, + date: new Date(2026, 0, 9), + }); + + await writer.write(meta()); + await writer.write(assistantText("first context")); + const firstPath = writer.currentPath(); + expect((await readTrace(firstPath)).length).toBe(2); + + await writer.rotate(); + const secondPath = writer.currentPath(); + expect(secondPath).not.toBe(firstPath); + expect(secondPath.endsWith(`${SESSION_ID}_002.jsonl`)).toBe(true); + + // The write goes into the new file. + await writer.write(assistantText("second context")); + expect((await readTrace(secondPath)).length).toBe(1); + + // append-only verification: the old file's row count does not increase. + expect((await readTrace(firstPath)).length).toBe(2); + }); +}); diff --git a/packages/core/test/validate-id.test.ts b/packages/core/test/validate-id.test.ts new file mode 100644 index 0000000..29f6730 --- /dev/null +++ b/packages/core/test/validate-id.test.ts @@ -0,0 +1,84 @@ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { assertValidId, isValidId, loadOrInitAgentState } from "../src/state/index.js"; + +describe("isValidId / assertValidId", () => { + const valid = ["default_project", "default_agent", "my-agent", "agent-1", "ABC_123"]; + // Only alphanumerics, `_`, and `-` are allowed; dots, spaces/tabs, path separators, and any + // other character are all rejected. + const invalid = [ + "", + " ", + "a b", + "a\tb", + " lead", + "trail ", + "a/b", + "a\\b", + ".", + "..", + "...", + "a.b", + "v1.0", + ".hidden", + "中文ok", + "emoji😀", + ]; + + it.each(valid)("接受合法单段目录名:%j", (id) => { + expect(isValidId(id)).toBe(true); + expect(() => assertValidId("project_id", id)).not.toThrow(); + expect(() => assertValidId("agent_id", id)).not.toThrow(); + }); + + it.each(invalid)("拒绝非法 id:%j", (id) => { + expect(isValidId(id)).toBe(false); + expect(() => assertValidId("agent_id", id)).toThrow(); + }); + + it("错误信息点明 kind 与非法 id", () => { + expect(() => assertValidId("agent_id", "a/b")).toThrow(/agent_id/); + expect(() => assertValidId("agent_id", "a/b")).toThrow(/a\/b/); + expect(() => assertValidId("project_id", "..")).toThrow(/project_id/); + }); +}); + +describe("loadOrInitAgentState id 校验", () => { + let tmpRoot: string; + let prevHome: string | undefined; + + beforeEach(async () => { + prevHome = process.env.PENGUIN_HOME; + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-validate-id-")); + process.env.PENGUIN_HOME = tmpRoot; + }); + + afterEach(async () => { + if (prevHome === undefined) { + delete process.env.PENGUIN_HOME; + } else { + process.env.PENGUIN_HOME = prevHome; + } + await fs.rm(tmpRoot, { recursive: true, force: true }); + }); + + it("agentId 含路径分隔符时抛错", async () => { + await expect(loadOrInitAgentState({ agentId: "a/b" })).rejects.toThrow(/agent_id/); + }); + + it("projectId 为 .. 时抛错", async () => { + await expect(loadOrInitAgentState({ projectId: ".." })).rejects.toThrow(/project_id/); + }); + + it("agentId 为 Windows 结尾空格变体 '.. ' 时抛错(防路径穿越绕过)", async () => { + await expect(loadOrInitAgentState({ agentId: ".. " })).rejects.toThrow(/agent_id/); + }); + + it("默认 id 合法,可正常初始化", async () => { + const state = await loadOrInitAgentState(); + expect(state.projectId).toBe("default_project"); + expect(state.agentId).toBe("default_agent"); + }); +}); diff --git a/packages/core/test/wake-signal.test.ts b/packages/core/test/wake-signal.test.ts new file mode 100644 index 0000000..a25853a --- /dev/null +++ b/packages/core/test/wake-signal.test.ts @@ -0,0 +1,33 @@ +/** + * Process-lifecycle behavior tests for WakeSignal. + */ +import { spawnSync } from "node:child_process"; +import { describe, expect, it } from "vitest"; + +describe("WakeSignal", () => { + it("keeps a standalone process alive while wait() is pending without leaving timers behind", () => { + const moduleUrl = new URL("../src/environment/tools/background/wake-signal.ts", import.meta.url) + .href; + const script = ` + const { WakeSignal } = await import(${JSON.stringify(moduleUrl)}); + + await new WakeSignal().wait(50); + process.stdout.write("timeout\\n"); + + const signal = new WakeSignal(); + setTimeout(() => signal.notify(), 10); + await signal.wait(60_000); + process.stdout.write("notified\\n"); + `; + + const result = spawnSync( + process.execPath, + ["--import", "tsx", "--input-type=module", "--eval", script], + { encoding: "utf8", timeout: 3_000 }, + ); + + expect(result.error).toBeUndefined(); + expect(result.status).toBe(0); + expect(result.stdout).toBe("timeout\nnotified\n"); + }); +}); diff --git a/packages/core/test/workspace.test.ts b/packages/core/test/workspace.test.ts new file mode 100644 index 0000000..e8d7c80 --- /dev/null +++ b/packages/core/test/workspace.test.ts @@ -0,0 +1,76 @@ +/** + * Temporary Workspace creation: + * + * - The directory name is the workspace_id, shaped like `tmp-<8hex>`, created under + * `/workspaces/`. + * - The id is checked for collisions within `workspaces/`: on conflict with an existing + * directory (EEXIST), it regenerates rather than reusing the existing directory; once + * retries are exhausted, it throws instead of silently falling back to an old temp Workspace. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { createTempWorkspace } from "../src/internal/session-support.js"; +import { workspacesDir } from "../src/state/paths.js"; + +const mocked = vi.hoisted(() => ({ uuids: [] as string[] })); + +vi.mock("node:crypto", async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + // When the queue has values, dequeue from it (tests inject fixed ids), otherwise fall back + // to the real implementation. + randomUUID: () => mocked.uuids.shift() ?? actual.randomUUID(), + }; +}); + +let tmpRoot: string; + +beforeEach(async () => { + tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-workspace-")); + mocked.uuids.length = 0; +}); + +afterEach(async () => { + await fs.rm(tmpRoot, { recursive: true, force: true }); +}); + +describe("createTempWorkspace", () => { + it("在 workspaces/ 下创建 tmp-<8hex> 目录并返回路径", async () => { + const dir = await createTempWorkspace(tmpRoot, "proj", "agent"); + expect(path.dirname(dir)).toBe(workspacesDir(tmpRoot, "proj", "agent")); + expect(path.basename(dir)).toMatch(/^tmp-[0-9a-f]{8}$/); + expect((await fs.stat(dir)).isDirectory()).toBe(true); + }); + + it("id 与已有目录冲突时重新生成,不复用已有目录", async () => { + const base = workspacesDir(tmpRoot, "proj", "agent"); + await fs.mkdir(path.join(base, "tmp-aaaaaaaa"), { recursive: true }); + await fs.writeFile(path.join(base, "tmp-aaaaaaaa", "keep.txt"), "old"); + mocked.uuids.push( + "aaaaaaaa-1111-4111-8111-111111111111", + "bbbbbbbb-2222-4222-8222-222222222222", + ); + + const dir = await createTempWorkspace(tmpRoot, "proj", "agent"); + + expect(path.basename(dir)).toBe("tmp-bbbbbbbb"); + expect(await fs.readdir(dir)).toEqual([]); + // The existing directory is neither reused nor modified. + await expect(fs.readFile(path.join(base, "tmp-aaaaaaaa", "keep.txt"), "utf8")).resolves.toBe( + "old", + ); + }); + + it("重试耗尽时报错,而非复用已有目录", async () => { + const base = workspacesDir(tmpRoot, "proj", "agent"); + await fs.mkdir(path.join(base, "tmp-cccccccc"), { recursive: true }); + mocked.uuids.push(...Array.from({ length: 16 }, () => "cccccccc-3333-4333-8333-333333333333")); + + await expect(createTempWorkspace(tmpRoot, "proj", "agent")).rejects.toThrow( + /unique temp workspace id/, + ); + }); +}); diff --git a/packages/core/tsconfig.json b/packages/core/tsconfig.json new file mode 100644 index 0000000..8cd1715 --- /dev/null +++ b/packages/core/tsconfig.json @@ -0,0 +1,7 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { + "rootDir": "." + }, + "include": ["src", "test"] +} diff --git a/packages/core/tsup.config.ts b/packages/core/tsup.config.ts new file mode 100644 index 0000000..c729da5 --- /dev/null +++ b/packages/core/tsup.config.ts @@ -0,0 +1,16 @@ +import { defineConfig } from "tsup"; + +export default defineConfig({ + // model-catalog gets its own entry point: pure data, no Node dependency, so web can bundle it directly via subpath import. + entry: [ + "src/index.ts", + "src/omnimessage/index.ts", + "src/interfaces.ts", + "src/state/model-catalog.ts", + ], + format: ["esm"], + target: "node20", + dts: true, + clean: true, + sourcemap: true, +}); diff --git a/packages/docs/README.md b/packages/docs/README.md new file mode 100644 index 0000000..77ae47e --- /dev/null +++ b/packages/docs/README.md @@ -0,0 +1,40 @@ +# @prismshadow/penguin-docs + +The PenguinHarness documentation site (React + Vite + Tailwind CSS 4). Its own package, sharing the landing page's visual language: bilingual pages (zh/en), light/dark themes, a sidebar + per-page table of contents, and a "Copy Markdown" button on every page. + +Live at . + +## Content model + +Pages are local Markdown files — `content/..md`, one file per page per language (missing languages fall back to the other). Frontmatter: + +```markdown +--- +title: Page title +description: One-line summary rendered under the title +--- +``` + +Navigation (sections, order, prev/next) is defined once in `src/lib/nav.ts`; a vitest check (`test/content.test.ts`) keeps nav and content in sync — every navigated slug must have both language files with a title, and no orphan files may exist. Internal links are written as absolute doc paths (`[Core Interfaces](/interfaces)`). + +## Development + +```bash +pnpm dev:docs # http://127.0.0.1:7367 (repo root script) +pnpm --filter @prismshadow/penguin-docs build # dist/ + per-route shells for deep links +pnpm --filter @prismshadow/penguin-docs typecheck +pnpm --filter @prismshadow/penguin-docs test +``` + +## Deployment + +Deployed to GitHub Pages together with the landing page as one artifact: `scripts/build-site.mjs` (repo root) builds landing with `BASE_PATH=` and docs with `BASE_PATH=docs/`, then copies this package's `dist/` into the landing dist under `docs/`. `.github/workflows/pages.yml` runs it on pushes to main touching either package. + +```bash +BASE_PATH=/ pnpm build:site # assemble locally +pnpm --filter @prismshadow/penguin-landing preview # serve the assembled site +``` + +Deep links work without a 404 fallback: the post-build step copies the SPA shell to `dist//index.html` for every content slug. + +Part of [PenguinHarness](https://github.com/Prism-Shadow/penguin-harness) · Apache-2.0 diff --git a/packages/docs/content/agent-loop.en.md b/packages/docs/content/agent-loop.en.md new file mode 100644 index 0000000..ab3aa91 --- /dev/null +++ b/packages/docs/content/agent-loop.en.md @@ -0,0 +1,123 @@ +--- +title: The Agent Loop +description: The context_engine's master flow diagram and a stage-by-stage breakdown — approvals, concurrent tool execution, interrupt carry-over, automatic reconnect and compaction. +--- + +The SDK's single execution entry point is `session.run(newMessages, opts?)`: input is the list of new OmniMessages (the Prompt); the return value is an async generator that streams [OmniMessage](/omni-message). One `run` drives one complete Task, until the model produces a final answer with no tool calls. + +This page shows the context_engine's overall flow first, then breaks down each stage; the message-level observable timeline and ordering guarantees are on [Message Flow & Ordering](/message-flow). Source: `packages/core/src/engine/context-engine.ts`. + +## The loop at a glance + +```text +session.run(newMessages, { approve, signal }) + │ carry-over from a previous interrupt? → prepend to this run's input + ▼ +┌── turn loop (≤ max_turns, default 100) ───────────────────────┐ +│ │ +│ request_begin │ +│ LLM.streamGenerate(newMessages) │ +│ ├─ streams partial_* fragments + complete msgs │ +│ ├─ for each complete tool_call: │ +│ │ approve(toolCall) ──deny──► synthetic aborted output │ +│ │ │allow (approvals sequential; │ +│ │ ▼ decision audited) │ +│ │ Environment.executeTool ──► runs concurrently, │ +│ │ output streams back │ +│ └─ LLMOutcome: │ +│ timeout / malformed ──► reconnect within the turn │ +│ (≤2, with ; tools not rerun) │ +│ token_usage + request_end (at LLM-stream end; not waiting │ +│ for tools) │ +│ │ +│ tool outputs reordered to original call order ──► next turn │ +│ no tool_call this turn? ──► Task ends, run returns │ +│ compaction trigger (context/turns)? ──► summarize/discard │ +│ + Trace rotation │ +└───────────────────────────────────────────────────────────────┘ + +signal fires (any point) ──► emit abort + build carry-over ──► run returns +``` + +Every message and event flows to two destinations at once: streamed live to the Human, and written to the [Trace](/sessions-and-traces). + +## Inputs and outputs + +```ts +const agent = await createAgent({ agentId: "default_agent" }); +const session = await agent.createSession({ workspaceDir: process.cwd() }); + +for await (const output of session.run([userText("Clean up the CSV files under data/")], { + approve: async (toolCall) => "allow", + signal: abortController.signal, +})) { + // output: partial_* fragments, complete model_msg, event_msg +} +``` + +```ts +interface RunOptions { + signal?: AbortSignal; // interrupt (e.g. Ctrl-C) + approve?: ApproveFn; // per-tool approval; denies everything when omitted (conservative default) +} +``` + +## Lifecycle of a turn + +A Task consists of consecutive Requests (turns). Each turn: + +1. emits `request_begin`; +2. the LLM streams back: `partial_*` fragments followed by complete messages; +3. every complete `tool_call` triggers exactly one `approve` callback; the decision is recorded as an `approval_decision` event; +4. approved calls run **concurrently** in the Environment (approvals themselves are one at a time); outputs stream out in completion order; +5. when the LLM stream ends, its final `token_usage` is emitted and `request_end(status)` follows at once — **without waiting for tools**: still-running tools may emit output after `request_end`; +6. once the whole batch is terminal, tool results are **reordered to the original call order** and become the next turn's input — the next Request never fires before that. + +The Task ends when a turn produces no `tool_call`. A denial produces a synthetic `aborted` tool output ("Tool call denied by user.") that the model reacts to. + +## Interruption and carry-over + +When `signal` fires, the engine emits an `abort` event and returns immediately, while constructing carry-over content for the next `run`: + +- **Case A — the model's output had completed** (the turn's `tool_call`s were committed): finished tool results are re-sent as structured `tool_call_output`s; unfinished calls get an `[interrupted: tool aborted by user]` placeholder, keeping `tool_call`/output pairing strictly intact; +- **Case B — the model's output was incomplete**: the whole turn is flattened into one `` user text carrying whatever partial output existed. + +Carry-over enters the model context only — it is never written to the Trace, which records only what actually happened. + +## Automatic reconnect + +Only LLM-side `timeout` (network timeouts, rate limits, 5xx) and `malformed` (truncated streams, JSON parse failures) trigger an in-run reconnect: the engine re-sends the original input plus a `` block carrying the previous partial output, so tools are never re-executed. Default limit is 2 reconnects with linear backoff (base 250ms); beyond that the turn settles as `failed`. Tool errors are never retried — they are fed back to the model as `tool_call_output` and the model decides what to do next. + +## Compaction + +Compaction settings are filled in from `system_config.yaml` by the composition layer: + +```ts +interface CompactionSettings { + maxContextLength: number; // context-token threshold (last token_usage's request.total); <=0 disables + maxSessionTurns: number; // cumulative Session turn threshold (counted across Tasks); <=0 = unlimited + mode: "summarize" | "discard"; + prompt: string; // the Prompt used by summarize compaction +} +``` + +Three triggers (`compaction_begin.reason`): + +| reason | Condition | +| --- | --- | +| `context` | last turn's `token_usage.request.total` ≥ `maxContextLength` (default 128000) | +| `turns` | Session turn count ≥ `maxSessionTurns` (default -1 = unlimited) | +| `manual` | the user runs `/compact` or calls `session.compact()` | + +Two modes: `summarize` (default) appends the compaction Prompt to the old context, extracts the ``, wraps it as a `` user text and continues in a **fresh model context**; `discard` simply drops the old context. Compaction rotates the [Trace file](/sessions-and-traces) (`_002`, `_003`, …) — one Trace file always equals one complete model context. `compactability()` probes feasibility before `session.compact()` (`ok | unsupported | empty | just_compacted`). + +## Concurrency model + +- Within a turn: approvals are sequential, execution is concurrent, and the next turn's input keeps the original order; +- within a Session: only one Task or one compaction runs at a time (the Server rejects concurrent requests with 409); +- a [Subagent](/tools) is an independent Session with its own Trace and loop; its messages are forwarded to the parent tagged with `origin`. + +## Side channels + +- **Session titles**: `session.generateTitle()` is a one-shot out-of-band LLM call (no tools, no system Prompt) that never enters history or Trace; +- **Usage accounting**: each turn's `token_usage` events are persisted row by row by the Server — the raw data behind the cost statistics. diff --git a/packages/docs/content/agent-loop.zh.md b/packages/docs/content/agent-loop.zh.md new file mode 100644 index 0000000..d6d5193 --- /dev/null +++ b/packages/docs/content/agent-loop.zh.md @@ -0,0 +1,120 @@ +--- +title: Agent 运行循环 +description: context_engine 的总体流程图与逐环节拆解——审批、并发工具执行、中断补发、自动重连与上下文压缩。 +--- + +SDK 的唯一执行入口是 `session.run(newMessages, opts?)`:输入本次新增的 OmniMessage 列表(Prompt),返回一个异步生成器,流式产出 [OmniMessage](/omni-message)。一次 `run` 自动跑完一个完整的 Task,直到模型给出不含工具调用的最终答复。 + +本页先给出 context_engine 的总体流程,再逐环节拆解;逐条消息级的可见时序与顺序保证见[消息流转与时序](/message-flow)。源码:`packages/core/src/engine/context-engine.ts`。 + +## 总体流程 + +```text +session.run(newMessages, { approve, signal }) + │ 存在上次中断的补发内容?→ 前置到本轮输入 + ▼ +┌── 轮循环(≤ max_turns,默认 100)──────────────────────────────┐ +│ │ +│ request_begin │ +│ LLM.streamGenerate(newMessages) │ +│ ├─ 流式产出 partial_* 分片 + 完整消息(thinking/text/…) │ +│ ├─ 每个完整 tool_call: │ +│ │ approve(toolCall) ──deny──► 合成 aborted 输出 │ +│ │ │allow (审批逐个;写审计事件) │ +│ │ ▼ │ +│ │ Environment.executeTool ──► 并发执行,输出流式回传 │ +│ └─ LLMOutcome: │ +│ timeout / malformed ──► 同轮自动重连(≤2 次, │ +│ 附 ,工具不重跑) │ +│ token_usage + request_end(LLM 流结束即产出,不等工具) │ +│ │ +│ 工具输出按原始调用顺序重排 ──► 作为下一轮输入 │ +│ 本轮无 tool_call?──► Task 结束,run 返回 │ +│ 压缩触发(context/turns)?──► summarize/discard + Trace 轮转 │ +└───────────────────────────────────────────────────────────────┘ + +signal 中断(任意时刻)──► 产出 abort 事件 + 构造补发内容 ──► run 返回 +``` + +全程的每条消息与事件同时流向两个去处:实时输出给 Human,以及写入 [Trace](/sessions-and-traces)。 + +## 输入与输出 + +```ts +const agent = await createAgent({ agentId: "default_agent" }); +const session = await agent.createSession({ workspaceDir: process.cwd() }); + +for await (const output of session.run([userText("整理 data/ 下的 CSV 文件")], { + approve: async (toolCall) => "allow", + signal: abortController.signal, +})) { + // output: partial_* 分片、完整 model_msg、event_msg +} +``` + +```ts +interface RunOptions { + signal?: AbortSignal; // 中断信号(如 Ctrl-C) + approve?: ApproveFn; // 逐工具审批;未注入时默认全部拒绝(保守策略) +} +``` + +## 一轮(Turn)的生命周期 + +Task 由若干连续的 Request(轮)组成,每轮: + +1. 产出 `request_begin`; +2. LLM 流式返回:`partial_*` 分片与完整消息依次产出; +3. 每个完整的 `tool_call` 恰好触发一次 `approve` 回调,决策以 `approval_decision` 事件记录; +4. 通过审批的工具交给 Environment **并发执行**(审批本身逐个进行),输出按完成顺序流出; +5. LLM 流结束时,先产出其最后一条 `token_usage`,随即产出 `request_end(status)`——**不等待工具**,仍在执行的工具输出可出现在 `request_end` 之后; +6. 整批工具全部到达终态后,工具结果**按原始调用顺序**重排,作为下一轮输入——在此之前不会发起下一次 Request。 + +某轮不再产生 `tool_call` 时,Task 结束。拒绝(deny)会生成一条合成的 `aborted` 工具输出(内容为 `Tool call denied by user.`),模型据此继续。 + +## 中断与补发(carry-over) + +`signal` 触发中断后,引擎产出 `abort` 事件并立即返回,同时为下一次 `run` 构造补发内容: + +- **场景 A:模型输出已完成**(该轮 `tool_call` 已提交)——已完成的工具结果按结构化 `tool_call_output` 补发;未执行完的调用补上 `[interrupted: tool aborted by user]` 占位,保证 `tool_call` 与输出严格配对; +- **场景 B:模型输出未完成**——整轮压平为一段 `` 用户文本,携带已产生的部分输出。 + +补发内容只进入模型上下文,不写入 Trace——Trace 永远只记录真实发生的消息。 + +## 自动重连 + +只有 LLM 侧的 `timeout`(网络超时、限流、5xx)与 `malformed`(流截断、JSON 解析失败)会触发引擎内自动重连:同一次 `run` 内重发原始输入,并附加 `` 块携带上一次的部分输出,避免工具重复执行。默认最多重连 2 次,线性退避(基数 250ms);超限后该轮以 `failed` 收场。工具错误从不重试——它们作为 `tool_call_output` 反馈给模型,由模型决定下一步。 + +## 上下文压缩(Compaction) + +压缩配置由组装层从 `system_config.yaml` 填充默认值: + +```ts +interface CompactionSettings { + maxContextLength: number; // 上下文 Token 阈值(取最近一次 token_usage 的 request.total);<=0 关闭 + maxSessionTurns: number; // Session 累计轮数阈值(跨 Task 计数);<=0 不限制 + mode: "summarize" | "discard"; + prompt: string; // summarize 模式使用的压缩 Prompt +} +``` + +三种触发方式(`compaction_begin.reason`): + +| reason | 触发条件 | +| --- | --- | +| `context` | 上一轮 `token_usage.request.total` ≥ `maxContextLength`(默认 128000) | +| `turns` | Session 轮数 ≥ `maxSessionTurns`(默认 -1,即不限) | +| `manual` | 用户执行 `/compact` 或调用 `session.compact()` | + +两种模式:`summarize`(默认)向旧上下文追加压缩 Prompt,提取 `` 后包装为 `` 用户文本,在**全新的模型上下文**中继续;`discard` 直接丢弃旧上下文。压缩时 [Trace 文件随之轮转](/sessions-and-traces)(`_002`、`_003`……),一个 Trace 文件恒等于一个完整模型上下文。`session.compact()` 前可用 `compactability()` 探询可行性(`ok | unsupported | empty | just_compacted`)。 + +## 并发模型 + +- 同一轮内:审批逐个、执行并发、下一轮输入按原始顺序; +- 同一 Session:同时只有一个 Task 或一次压缩在运行(Server 侧以 409 拒绝并发请求); +- [Subagent](/tools) 是独立 Session,拥有自己的 Trace 与运行循环,消息以 `origin` 标记转发给父级。 + +## 相关旁路 + +- **Session 标题**:`session.generateTitle()` 走独立的一次性 LLM 调用(无工具、无系统 Prompt),不进入历史与 Trace; +- **用量落账**:每轮的 `token_usage` 事件被 Server 逐条入库,构成成本统计的原始数据。 diff --git a/packages/docs/content/architecture.en.md b/packages/docs/content/architecture.en.md new file mode 100644 index 0000000..6ffa1af --- /dev/null +++ b/packages/docs/content/architecture.en.md @@ -0,0 +1,135 @@ +--- +title: Architecture +description: How the three-interface boundary, the context_engine and OmniMessage organize the SDK, CLI, Server and Web App into one system. +--- + +PenguinHarness is a pnpm monorepo whose center is the execution engine in `@prismshadow/penguin-core`; the CLI, the Server and the Web App are just different "Human implementations" of that same engine. + +## Layers + +```text +┌─────────────┐ ┌─────────────────────────────┐ +│ CLI │ │ Web App (React SPA) │ +│ (penguin) │ │ ↑ OmniMessage over SSE │ +│ │ │ Server (Hono + SQLite) │ +└──────┬──────┘ └──────────────┬──────────────┘ + │ session.run(...) │ ← Human boundary +┌──────┴────────────────────────┴──────────────┐ +│ core: context_engine (ReAct loop) │ +│ ├── LLMInterface ──→ AgentHub ──→ models │ +│ ├── EnvironmentInterface ──→ builtin tools│ +│ ├── Agent State (editable files) │ +│ └── Trace (append-only JSONL) │ +└──────────────────────────────────────────────┘ +``` + +| Package | Role | +| --- | --- | +| `packages/core` | SDK and engine: context_engine, OmniMessage, LLM/Environment interfaces, State and Trace | +| `packages/cli` | Terminal Human implementation: REPL and one-shot runs, embeds core in-process | +| `packages/server` | Web Human implementation: HTTP for input and approvals, SSE for the output stream | +| `packages/web` | Rendering SPA: streams by the OmniMessage protocol, contains no engine logic | +| `packages/skills` | The built-in skill library (a set of `SKILL.md` files) | + +## Division of responsibilities + +To place a design in a layer, ask where its **source of truth** lives. The four layers split as: + +| Layer | Owns | Does not own | +| --- | --- | --- | +| SDK (`core`) | Protocol and execution — everything that makes messages flow | persisted user state, multi-user, any rendering | +| Server | The resident process and the multi-user runtime | engine logic (fully delegated to the SDK) | +| File layer (`~/.penguin/data`) | Everything editable and everything recorded | any computation | +| CLI / Web | Rendering and interaction | business state | + +Item by item (design → owner → carrying file or module): + +| Design | Owner | Carried by | +| --- | --- | --- | +| The OmniMessage protocol, message parsing and partial aggregation | SDK | `core/src/omnimessage/` — see [The OmniMessage Protocol](/omni-message) | +| The ReAct loop, carry-over, reconnect, compaction | SDK | `core/src/engine/context-engine.ts` — see [The Agent Loop](/agent-loop) | +| The approval mechanism (one decision per tool_call) | SDK | `ApproveFn` (`core/src/interfaces.ts`); the concrete mode is injected by CLI/Server | +| Tool execution and centralized close-out | SDK | `core/src/environment/` — see [Tools & Approval](/tools) | +| Model access (provider protocol adaptation) | SDK → AgentHub | `core/src/llm/` + `@prismshadow/agenthub` — see [Models & Providers](/models) | +| Trace writing and Session-recovery logic | SDK | `core/src/trace/` (the records themselves live in the file layer) | +| Subagent spawning and message forwarding | SDK | the `run_subagent` tool + the injected `SubagentRunner` | +| Multi-user auth and Project authorization | Server | `server/src/auth/`, `server/src/services/project-service.ts` | +| Session indexing, per-Session mutex, SSE forwarding | Server | `server/src/runtime/` — see [Server API](/server-api) | +| Scheduled tasks (execution) | Server | `server/src/runtime/scheduler.ts`; the task definitions live in the file layer at `agent_state/schedule/*.toml` | +| Approval-mode persistence and manual decisions | Server | `server/src/runtime/approvals.ts` + SQLite | +| Usage persistence and cost statistics | Server | `server/src/runtime/usage-recorder.ts`, `services/usage-service.ts` | +| Agent behavior definition (prompts, runtime params) | File layer | `agent_state/system_config.yaml`, `AGENTS.md` — see the [Configuration Reference](/configuration) | +| Skills | File layer | `agent_state/skills//SKILL.md` — see [Skills](/skills) | +| Secrets | File layer | Vault: `agent_state/.vault.toml`; model credentials: `.project_config.toml` (both 0600) | +| The model table and the default model | File layer | `/.project_config.toml` | +| Run history (the sole source of truth for recovery) | File layer | `traces//_.jsonl` — see [Sessions & Traces](/sessions-and-traces) | +| Benchmark cases and scores | File layer | `benchmarks//` — see [Self-Improvement](/self-improvement) | +| Snapshots | File layer | `snapshots/v.tar.gz`; the export/import service lives in the Server | +| Streaming rendering, approval UI, charts | CLI / Web | `cli/src`, `web/src` (pure rendering, no engine logic) | + +The one-line rule: **what is editable or recorded lives in files; what makes messages flow lives in the SDK; what needs a resident process and multiple users lives in the Server; the rest is rendering.** The Server's SQLite stores only indexes and aggregates — it never competes with the file layer as a source of truth. + +## Source layout + +How each package is organized (single-purpose files, split by layer; every file's header comment is its design note): + +```text +packages/ +├── core/src +│ ├── agent.ts / session.ts # the createAgent composition layer and Session (run / compact / generateTitle) +│ ├── session-title.ts # one-shot title generation (out-of-band LLM call, never in Trace) +│ ├── engine/context-engine.ts # ReAct loop orchestration: turn lifecycle, approvals, carry-over, reconnect, compaction +│ ├── omnimessage/ # types.ts protocol types · builders.ts constructors · aggregate.ts partial aggregation +│ ├── llm/ # generative-model.ts AgentHub adapter · tool-call-ids.ts id uniqueness +│ ├── environment/ # environment.ts execution close-out · tools/ registry, 6 builtin tools, background sessions +│ ├── state/ # paths · default-config · project-config · model-catalog +│ │ # agent-state (Skill install, prompt assembly) · agent-vault · builtin-agents +│ ├── trace/ # writer.ts append-only JSONL · resume.ts replay-based recovery +│ └── internal/ # date and Session helpers +├── cli/src # commander entry + run / chat / config / serve commands and approval prompts +├── server/src # app assembly · db (node:sqlite) · auth · http/routes · runtime · services +├── web/src # api client · state · lib/omni stream rendering · components · feature pages +├── skills/ # loader + the skills//SKILL.md library +├── landing/ # the product landing page (with the blog) +└── docs/ # this documentation site +``` + +The internals of server and web are detailed on [Server API](/server-api) and the [Web App Guide](/web-app). + +## The three-interface boundary + +The context_engine is the heart of the system and does exactly two things: it maintains the linear message history, and it orchestrates the event flow between three interfaces. It speaks only [OmniMessage](/omni-message) and performs no protocol conversion: + +- **Human** — the user-side boundary. It is deliberately not an interface class: the SDK's single entry point `session.run(newMessages, { approve, signal })` *is* the Human boundary. Input is a list of new OmniMessages plus an approval callback; output is streamed OmniMessages. The CLI and the Server are its two shipped implementations. +- **LLM** — the model-side interface (`LLMInterface`). Translates OmniMessage to requests against the AgentHub model gateway and streamed events back into OmniMessage. All provider protocol adaptation happens inside AgentHub; core never imports a vendor SDK. +- **Environment** — the tool-execution interface (`EnvironmentInterface`). Runs approved tool calls and streams results back. + +Why this boundary matters: the kernel contains no provider, tool or UI specifics, so each side swaps by configuration (local shell today, other sandboxes tomorrow; CLI, Web, or programmatic callers) without touching the core. See [Core Interfaces](/interfaces) for the signatures. + +## Data flow of one Task + +1. Human hands a Prompt (a list of OmniMessages) to `session.run`; +2. the engine issues a Request: the LLMInterface streams `partial_*` fragments and complete messages; +3. every complete `tool_call` triggers one `approve` decision; approved calls run concurrently in the Environment; +4. tool outputs are re-fed in their original order as the next Request's input; +5. the Task ends when a turn produces no `tool_call` (the final answer). + +Every message and event flows to two destinations at once: streamed live to the Human, and appended to the [Trace](/sessions-and-traces). Loop details (interruption, reconnect, compaction) are on [The Agent Loop](/agent-loop). + +## The state layer + +Below the engine sits a purely file-based state layer rooted at `~/.penguin/data` (override with `PENGUIN_HOME`), organized as `/agents//`: + +- **Agent State** — the `agent_state/` directory: `system_config.yaml`, `AGENTS.md`, Skills, Vault. An Agent's entire behavior is editable files. +- **Project config** — `.project_config.toml`: the model table and credentials; model identity is always the `(provider, model_id)` pair. +- **Trace** — the `traces/` directory: append-only JSONL, the single source of truth for Session recovery. + +The Server keeps an additional SQLite index (users, authorization, usage stats) but never duplicates the file layer's facts — the CLI, SDK and Web share one data directory and can be mixed freely. + +## Key design decisions + +- **One protocol, three jobs**: OmniMessage is simultaneously the SDK's external interface, the Trace on-disk format and the engine's internal currency — what streams, what is stored and what the model sees are the same thing. +- **Errors converge into messages**: the LLM and Environment never throw into the engine; results carry a five-value `stop_reason` (`completed | failed | aborted | timeout | malformed`), and only LLM-side `timeout / malformed` trigger an in-run reconnect. +- **A thin model layer**: core defines only `LLMInterface`; provider adaptation lives entirely in AgentHub (`@prismshadow/agenthub`), which is what makes any OpenAI-compatible endpoint reachable. See [Models & Providers](/models). + +Source entry points: `packages/core/src/engine/context-engine.ts`, `packages/core/src/interfaces.ts`. diff --git a/packages/docs/content/architecture.zh.md b/packages/docs/content/architecture.zh.md new file mode 100644 index 0000000..0824a83 --- /dev/null +++ b/packages/docs/content/architecture.zh.md @@ -0,0 +1,135 @@ +--- +title: 架构总览 +description: 三接口边界、context_engine 与 OmniMessage 如何把 SDK、CLI、Server、Web 组织成一个系统。 +--- + +PenguinHarness 是一个 pnpm monorepo,核心是 `@prismshadow/penguin-core` 中的执行引擎;CLI、Server 与 Web App 都只是这同一个引擎的不同「Human 实现」。 + +## 分层结构 + +```text +┌─────────────┐ ┌─────────────────────────────┐ +│ CLI │ │ Web App (React SPA) │ +│ (penguin) │ │ ↑ OmniMessage over SSE │ +│ │ │ Server (Hono + SQLite) │ +└──────┬──────┘ └──────────────┬──────────────┘ + │ session.run(...) │ ← Human 边界 +┌──────┴────────────────────────┴──────────────┐ +│ core: context_engine(ReAct 循环) │ +│ ├── LLMInterface ──→ AgentHub ──→ 各模型 │ +│ ├── EnvironmentInterface ──→ 内置工具 │ +│ ├── Agent State(可编辑文件) │ +│ └── Trace(追加式 JSONL) │ +└──────────────────────────────────────────────┘ +``` + +| 包 | 角色 | +| --- | --- | +| `packages/core` | SDK 与执行引擎:context_engine、OmniMessage、LLM/Environment 接口、State 与 Trace | +| `packages/cli` | 终端 Human 实现:REPL 与单次运行,直接内嵌 core | +| `packages/server` | Web Human 实现:HTTP 承接输入与审批,SSE 推送输出流 | +| `packages/web` | 渲染层 SPA:按 OmniMessage 协议流式渲染,不含业务引擎 | +| `packages/skills` | 内置技能库(`SKILL.md` 文件集合) | + +## 职责划分 + +判断一个设计归属哪一层,只看它的**事实来源**在哪里。四层的分工: + +| 层 | 承担 | 不承担 | +| --- | --- | --- | +| SDK(`core`) | 协议与执行:让消息流动起来的一切 | 持久化用户态、多用户、任何渲染 | +| Server | 常驻进程与多用户运行时 | 引擎逻辑(全部委托给 SDK) | +| 文件层(`~/.penguin/data`) | 一切可编辑的定义与一切被记录的历史 | 任何计算 | +| CLI / Web | 渲染与交互 | 业务状态 | + +逐项对应(设计 → 归属 → 承载文件或模块): + +| 设计 | 归属 | 承载 | +| --- | --- | --- | +| OmniMessage 协议、消息解析与分片聚合 | SDK | `core/src/omnimessage/`,见 [OmniMessage 协议](/omni-message) | +| ReAct 循环、补发、重连、压缩 | SDK | `core/src/engine/context-engine.ts`,见 [Agent 运行循环](/agent-loop) | +| 审批机制(每个 tool_call 一次决策) | SDK | `ApproveFn`(`core/src/interfaces.ts`);具体模式由 CLI/Server 注入 | +| 工具执行与统一收尾 | SDK | `core/src/environment/`,见[工具与审批](/tools) | +| 模型接入(Provider 协议适配) | SDK → AgentHub | `core/src/llm/` + `@prismshadow/agenthub`,见[模型与 Provider](/models) | +| Trace 写入与 Session 恢复逻辑 | SDK | `core/src/trace/`(记录本体在文件层) | +| Subagent 派生与消息回流 | SDK | `run_subagent` 工具 + `SubagentRunner` 注入 | +| 多用户认证与 Project 授权 | Server | `server/src/auth/`、`server/src/services/project-service.ts` | +| Session 索引、并发互斥、SSE 转发 | Server | `server/src/runtime/`,见 [Server API](/server-api) | +| 定时任务(Schedule 执行) | Server | `server/src/runtime/scheduler.ts`;任务定义在文件层 `agent_state/schedule/*.toml` | +| 审批模式持久化与人工决策 | Server | `server/src/runtime/approvals.ts` + SQLite | +| 用量落库与成本统计 | Server | `server/src/runtime/usage-recorder.ts`、`services/usage-service.ts` | +| Agent 行为定义(Prompt、运行参数) | 文件层 | `agent_state/system_config.yaml`、`AGENTS.md`,见[配置参考](/configuration) | +| Skill | 文件层 | `agent_state/skills//SKILL.md`,见[技能系统](/skills) | +| 密钥 | 文件层 | Vault:`agent_state/.vault.toml`;模型凭据:`.project_config.toml`(均 0600) | +| 模型表与默认模型 | 文件层 | `/.project_config.toml` | +| 运行历史(恢复的唯一事实来源) | 文件层 | `traces//_.jsonl`,见 [Session 与 Trace](/sessions-and-traces) | +| Benchmark 题库与评分 | 文件层 | `benchmarks//`,见[自我进化](/self-improvement) | +| 快照 | 文件层 | `snapshots/v.tar.gz`;导入导出服务在 Server | +| 流式渲染、审批 UI、统计图表 | CLI / Web | `cli/src`、`web/src`(纯渲染,不含引擎逻辑) | + +一句话判定:**能编辑的与被记录的在文件层;让消息流动起来的在 SDK;需要常驻进程与多用户的在 Server;其余是渲染。**Server 的 SQLite 只存索引与聚合,从不与文件层争当事实来源。 + +## 源码结构 + +各包的目录设计(职责单一、按层拆分;文件头注释即该文件的设计说明): + +```text +packages/ +├── core/src +│ ├── agent.ts / session.ts # createAgent 组装层与 Session(run / compact / generateTitle) +│ ├── session-title.ts # 一次性标题生成(旁路 LLM 调用,不入 Trace) +│ ├── engine/context-engine.ts # ReAct 循环编排:轮生命周期、审批、补发、重连、压缩 +│ ├── omnimessage/ # types.ts 协议类型 · builders.ts 构造函数 · aggregate.ts 分片聚合 +│ ├── llm/ # generative-model.ts AgentHub 适配 · tool-call-ids.ts id 唯一化 +│ ├── environment/ # environment.ts 执行与收尾 · tools/ 注册表、6 个内置工具、后台会话 +│ ├── state/ # paths · default-config · project-config · model-catalog +│ │ # agent-state(Skill 安装、提示词装配)· agent-vault · builtin-agents +│ ├── trace/ # writer.ts 追加式 JSONL · resume.ts 回放恢复 +│ └── internal/ # 日期与 Session 辅助 +├── cli/src # commander 入口 + run / chat / config / serve 命令与审批交互 +├── server/src # app 组装 · db(node:sqlite)· auth · http/routes · runtime · services +├── web/src # api 客户端 · state · lib/omni 流渲染 · components · features 各页面 +├── skills/ # 加载器 + skills//SKILL.md 技能库 +├── landing/ # 产品落地页(含博客) +└── docs/ # 本文档站 +``` + +server 与 web 的内部结构分别见 [Server API](/server-api) 与 [Web App 指南](/web-app)。 + +## 三接口边界 + +context_engine 是整个系统的核心,它只做两件事:维护线性消息历史,以及在三个接口之间编排事件流。它只认识 [OmniMessage](/omni-message),不做任何协议转换: + +- **Human**——用户侧边界。它不是一个接口类:SDK 的唯一入口 `session.run(newMessages, { approve, signal })` 就是 Human 边界本身。输入是新增的 OmniMessage 列表与审批回调,输出是流式 OmniMessage。CLI 与 Server 是它的两种实现形态。 +- **LLM**——模型侧接口(`LLMInterface`)。把 OmniMessage 翻译为模型网关 AgentHub 的请求,把流式事件翻译回 OmniMessage。所有 Provider 协议适配都在 AgentHub 内完成,core 不直接依赖任何模型厂商 SDK。 +- **Environment**——工具执行接口(`EnvironmentInterface`)。执行通过审批的工具调用,把结果以流式 OmniMessage 送回。 + +这一边界设计的意义:引擎内核不含任何 Provider、工具或 UI 细节,三侧实现均可按配置替换(本地 shell、其他执行沙箱;CLI、Web、程序化调用),而互不影响。接口签名详见[接口契约](/interfaces)。 + +## 一个 Task 的数据流 + +1. Human 把 Prompt(OmniMessage 列表)交给 `session.run`; +2. 引擎发起一次 Request:经 LLMInterface 流式产出 `partial_*` 与完整消息; +3. 每个完整的 `tool_call` 触发一次 `approve` 审批;通过后交 Environment 并发执行; +4. 工具输出按原始顺序回填,进入下一轮 Request; +5. 某轮不再产生 `tool_call`(最终答复)时 Task 结束。 + +全程的每条消息与事件同时流向两个去处:实时输出给 Human,以及追加写入 [Trace](/sessions-and-traces)。运行循环的细节(中断、重连、压缩)见 [Agent 运行循环](/agent-loop)。 + +## 状态层 + +引擎之下是纯文件的状态层,数据根目录为 `~/.penguin/data`(`PENGUIN_HOME` 可改),按 `/agents//` 组织: + +- **Agent State**——`agent_state/` 目录:`system_config.yaml`、`AGENTS.md`、Skills、Vault。Agent 的全部行为定义都是可编辑文件。 +- **Project 配置**——`.project_config.toml`:模型表与凭据,模型身份恒为 `(provider, model_id)` 二元组。 +- **Trace**——`traces/` 目录:追加式 JSONL,恢复 Session 的唯一事实来源。 + +Server 额外维护一个 SQLite 索引库(用户、授权、用量统计),但从不复制文件层的事实——CLI、SDK 与 Web 共用同一份数据目录,可以混用。 + +## 关键设计决策 + +- **一个协议,三种职责**:OmniMessage 同时是 SDK 对外接口、Trace 落盘格式与引擎内部通货——「流出去的」「存下来的」「模型看到的」是同一种东西。 +- **错误收敛为消息**:LLM 与 Environment 从不向引擎抛异常;结果携带五值 `stop_reason`(`completed | failed | aborted | timeout | malformed`),仅 LLM 侧的 `timeout / malformed` 触发引擎内重连。 +- **薄模型层**:core 只定义 `LLMInterface`,Provider 适配全部下沉到 AgentHub(`@prismshadow/agenthub`),因此支持任意 OpenAI 兼容端点,见[模型与 Provider](/models)。 + +源码入口:`packages/core/src/engine/context-engine.ts`、`packages/core/src/interfaces.ts`。 diff --git a/packages/docs/content/cli.en.md b/packages/docs/content/cli.en.md new file mode 100644 index 0000000..6e94071 --- /dev/null +++ b/packages/docs/content/cli.en.md @@ -0,0 +1,145 @@ +--- +title: CLI Reference +description: Complete reference for the penguin command, its subcommands, and options. +--- + +The CLI ships as the npm package `@prismshadow/penguin-cli`; the command is `penguin`. Running bare `penguin` prints help; `-v, --version` prints the version. A `.env` file in the working directory is loaded automatically on startup. + +## Global conventions + +- Model references: a model's identity is always the `(provider, model_id)` pair. `--model-id` takes the upstream model id and pairs with `--provider`. When `run` / `chat` omit `--provider`, the `--model-id` matches only if it is globally unique in the configuration; ambiguity is an error. +- Data root: `--root ` overrides the data root directory. Priority: `--root` > the `PENGUIN_HOME` env var > `~/.penguin/data`. + +## penguin run + +Send a single message, execute one Task, then exit. If the Task aborted, the exit code is non-zero, so scripts / CI can check it. + +```bash +penguin run -m "Summarize the code structure of this directory" +``` + +| Option | Description | +| --- | --- | +| `-m, --message ` | Required; the message to send | +| `--model-id ` | Model to use; defaults to the Project's default model | +| `--provider ` | Provider group of the model | +| `--project-id ` | Project to use | +| `--agent-id ` | Agent to use | +| `--workspace ` | Workspace directory; defaults to the current directory and must exist | +| `--approve ` | Approval mode, see below | + +## penguin chat + +Interactive REPL; each input line starts a Task. Takes the same options as `run` (minus `-m, --message`), plus: + +| Option | Description | +| --- | --- | +| `--resume [sessionId]` | Resume a Session; without an id, resumes the Agent's latest Session | + +With `--resume`, the Workspace and model are locked by the original Session and cannot be overridden via `--workspace` / `--model-id` / `--provider`. On exit, a copy-pastable `penguin chat --resume ` command is printed. + +In-REPL commands: + +| Input | Behavior | +| --- | --- | +| `/compact` | Proactively compact the current context | +| `/exit`, `/quit` | Quit | + +Ctrl-C is state-dependent: + +| State | Behavior | +| --- | --- | +| Awaiting tool approval | Deny that tool call | +| Task running | Abort the current Task and return to input | +| Input buffer non-empty | Clear the current input | +| Idle with empty buffer | Show an exit confirmation (y/N) | + +## Approval modes (--approve) + +| Mode | Behavior | +| --- | --- | +| `allow-all` | Auto-approve every tool call (default) | +| `deny-all` | Auto-reject every tool call | +| `read-only` | Auto-approve read-only tools; prompt for the rest | +| `always-ask` | Prompt for every tool call | + +At an interactive prompt, `y` / `yes` approves and `n` / `no` denies; a bare Enter defaults to approve. + +## penguin config + +Manages a Project's model configuration, per-Agent vault environment variables, and the UI language. Except for `lang`, all subcommands below accept `--project-id ` (defaults to the default Project) and `--root `. + +### model add + +Add or update a model entry: + +```bash +penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default +``` + +| Option | Description | +| --- | --- | +| `--model-id ` | Required; the upstream model id | +| `--provider ` | Provider group; inferred from the built-in catalog when omitted | +| `--api-key ` | API key, stored inline in the Project's hidden `.project_config.toml` | +| `--base-url ` | Custom endpoint base URL | +| `--context-window ` | Context window size | +| `--client-type ` | Client protocol type | +| `--vision` / `--no-vision` | Mark vision input as supported / unsupported | +| `--price-cache-read ` | Cache-read price | +| `--price-cache-write ` | Cache-write price | +| `--price-output ` | Output price | +| `--set-default` | Also set as the default model | + +### model default / model vision / model list + +```bash +penguin config model default --model-id --provider +penguin config model vision --model-id --provider +penguin config model list +``` + +- `model default` sets the Project's default model; `model vision` sets the vision proxy model. Both require `--model-id` and `--provider`, and the reference must already exist in the model list. +- `model list` lists configured models; the default model is marked with `*`. + +### vault + +Per-Agent environment variable store, written to `agent_state/.vault.toml`. Values are injected into tool subprocess environments only — never into the model context. + +```bash +penguin config vault set --key GITHUB_TOKEN --value ghp_xxx +penguin config vault list +penguin config vault remove --key GITHUB_TOKEN +``` + +| Subcommand | Options | +| --- | --- | +| `vault set` | `--key ` (required), `--value ` (required), `[--agent-id ]` | +| `vault list` | `[--agent-id ]` | +| `vault remove` | `--key ` (required), `[--agent-id ]` | + +### lang + +```bash +penguin config lang en +``` + +Sets the CLI UI language (`en` or `zh`) by writing `PENGUIN_LANG` into the shell startup file. + +## penguin server / penguin web + +Two entry points into the same service process: `server` runs headless; `web` additionally waits for readiness, prints the URL, and opens the browser. + +```bash +penguin web +``` + +| Option | Description | +| --- | --- | +| `--port ` | Listen port, default 7364 | +| `--host ` | Listen host, default 127.0.0.1 | +| `--no-open` | `web` only: do not open the browser | + +Port / host priority: command-line option > the `PORT` / `HOST` env vars (including `.env`) > defaults. + +See also: [Configuration Reference](/configuration), [Models & Providers](/models). diff --git a/packages/docs/content/cli.zh.md b/packages/docs/content/cli.zh.md new file mode 100644 index 0000000..2241400 --- /dev/null +++ b/packages/docs/content/cli.zh.md @@ -0,0 +1,145 @@ +--- +title: CLI 参考 +description: penguin 命令的子命令与选项完整参考。 +--- + +CLI 由 npm 包 `@prismshadow/penguin-cli` 提供,命令为 `penguin`。不带子命令执行 `penguin` 时打印帮助;`-v, --version` 打印版本号。启动时自动加载工作目录下的 `.env`。 + +## 全局约定 + +- 模型引用:模型身份始终是 `(provider, model_id)` 二元组。`--model-id` 填上游模型 id,与 `--provider` 组成配对引用。`run` / `chat` 省略 `--provider` 时,仅当该 `--model-id` 在配置中全局唯一才会匹配,存在歧义则报错。 +- 数据根目录:`--root ` 覆盖数据根目录,优先级为 `--root` > 环境变量 `PENGUIN_HOME` > `~/.penguin/data`。 + +## penguin run + +发送单条消息执行一个 Task,结束后退出;Task 被中止时以非零码退出,便于脚本 / CI 判断。 + +```bash +penguin run -m "总结当前目录的代码结构" +``` + +| 选项 | 说明 | +| --- | --- | +| `-m, --message ` | 必填,要发送的消息 | +| `--model-id ` | 指定模型,缺省使用 Project 默认模型 | +| `--provider ` | 模型所属 Provider 分组 | +| `--project-id ` | 指定 Project | +| `--agent-id ` | 指定 Agent | +| `--workspace ` | Workspace 目录,默认当前目录,必须已存在 | +| `--approve ` | 审批模式,见下文 | + +## penguin chat + +交互式 REPL,每输入一行发起一个 Task。选项与 `run` 相同(除 `-m, --message` 外),另加: + +| 选项 | 说明 | +| --- | --- | +| `--resume [sessionId]` | 恢复指定 Session;省略 id 时恢复该 Agent 最近的 Session | + +使用 `--resume` 时,Workspace 与模型由原 Session 锁定,不可再用 `--workspace` / `--model-id` / `--provider` 覆盖。退出时会打印可直接复制的 `penguin chat --resume ` 命令。 + +REPL 内命令: + +| 输入 | 行为 | +| --- | --- | +| `/compact` | 主动压缩当前上下文 | +| `/exit`、`/quit` | 退出 | + +Ctrl-C 的行为依状态而定: + +| 状态 | 行为 | +| --- | --- | +| 等待工具审批 | 拒绝该次工具调用 | +| Task 运行中 | 中断当前 Task,返回输入 | +| 输入缓冲非空 | 清空当前输入 | +| 空闲且缓冲为空 | 显示退出确认(y/N) | + +## 审批模式(--approve) + +| 模式 | 行为 | +| --- | --- | +| `allow-all` | 自动批准所有工具调用(默认) | +| `deny-all` | 自动拒绝所有工具调用 | +| `read-only` | 自动批准只读工具,其余逐个询问 | +| `always-ask` | 每次工具调用都询问 | + +交互询问时输入 `y` / `yes` 批准、`n` / `no` 拒绝;直接回车默认为批准。 + +## penguin config + +管理 Project 的模型配置、Agent 级 vault 环境变量与界面语言。除 `lang` 外,以下子命令均支持 `--project-id `(缺省为默认 Project)与 `--root `。 + +### model add + +新增或更新模型条目: + +```bash +penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default +``` + +| 选项 | 说明 | +| --- | --- | +| `--model-id ` | 必填,上游模型 id | +| `--provider ` | Provider 分组,缺省时根据内置目录推断 | +| `--api-key ` | API Key,内联存入 Project 隐藏文件 `.project_config.toml` | +| `--base-url ` | 自定义接口地址 | +| `--context-window ` | 上下文窗口大小 | +| `--client-type ` | 客户端协议类型 | +| `--vision` / `--no-vision` | 标记是否支持视觉输入 | +| `--price-cache-read ` | 缓存读价格 | +| `--price-cache-write ` | 缓存写价格 | +| `--price-output ` | 输出价格 | +| `--set-default` | 同时设为默认模型 | + +### model default / model vision / model list + +```bash +penguin config model default --model-id --provider +penguin config model vision --model-id --provider +penguin config model list +``` + +- `model default` 设置 Project 默认模型;`model vision` 设置视觉代理模型。两者的 `--model-id` 与 `--provider` 均为必填,且引用必须已存在于模型列表。 +- `model list` 列出已配置模型,默认模型以 `*` 标记。 + +### vault + +按 Agent 存储环境变量,写入 `agent_state/.vault.toml`;值只注入工具子进程的环境变量,绝不进入模型上下文。 + +```bash +penguin config vault set --key GITHUB_TOKEN --value ghp_xxx +penguin config vault list +penguin config vault remove --key GITHUB_TOKEN +``` + +| 子命令 | 选项 | +| --- | --- | +| `vault set` | `--key `(必填)、`--value `(必填)、`[--agent-id ]` | +| `vault list` | `[--agent-id ]` | +| `vault remove` | `--key `(必填)、`[--agent-id ]` | + +### lang + +```bash +penguin config lang zh +``` + +设置 CLI 界面语言(`en` 或 `zh`),将 `PENGUIN_LANG` 写入 shell 启动文件。 + +## penguin server / penguin web + +两者是同一服务进程的两个入口:`server` 为 headless 模式;`web` 额外等待服务就绪、打印 URL 并打开浏览器。 + +```bash +penguin web +``` + +| 选项 | 说明 | +| --- | --- | +| `--port ` | 监听端口,默认 7364 | +| `--host ` | 监听地址,默认 127.0.0.1 | +| `--no-open` | 仅 `web`:不自动打开浏览器 | + +端口 / 地址优先级:命令行选项 > 环境变量 `PORT` / `HOST`(含 `.env`)> 默认值。 + +相关文档:[配置参考](/configuration)、[模型与 Provider](/models)。 diff --git a/packages/docs/content/configuration.en.md b/packages/docs/content/configuration.en.md new file mode 100644 index 0000000..909b049 --- /dev/null +++ b/packages/docs/content/configuration.en.md @@ -0,0 +1,186 @@ +--- +title: Configuration Reference +description: Complete field reference for environment variables, Project config, Agent config, the Vault, and Schedules. +--- + +PenguinHarness configuration has three layers: environment variables shape the deployment, the Project config manages models and credentials, and the Agent config defines a single Agent's behavior. Each Agent additionally has two kinds of state files: the Vault (private environment variables) and Schedules (timed tasks). + +## Environment variables + +The CLI and the server automatically load a `.env` file from the working directory on startup. + +| Variable | Description | Default | +| --- | --- | --- | +| `PENGUIN_HOME` | Data root directory | `~/.penguin/data` | +| `PORT` | Web service listen port | `7364` | +| `HOST` | Web service listen address | `127.0.0.1` | +| `PENGUIN_WEB_DB` | Server SQLite database path | `/web.db` | +| `PENGUIN_WEB_DIST` | Front-end static assets directory | the npm server package falls back to its bundled web-dist | +| `PENGUIN_LANG` | CLI language (`en` / `zh`), set via `penguin config lang` | `en` | + +### Provider credential variables + +When a model entry has no inline `api_key`, the AgentHub gateway falls back to the provider's environment variable; the `*_BASE_URL` variants override the base URL the same way: + +| Provider | API key | Base URL | +| --- | --- | --- | +| deepseek | `DEEPSEEK_API_KEY` | `DEEPSEEK_BASE_URL` | +| anthropic | `ANTHROPIC_API_KEY` | `ANTHROPIC_BASE_URL` | +| openai, openrouter, siliconflow, custom | `OPENAI_API_KEY` | `OPENAI_BASE_URL` | +| google | `GEMINI_API_KEY` | `GEMINI_BASE_URL` | +| zhipu | `ZAI_API_KEY` | `ZAI_BASE_URL` | +| moonshot | `MOONSHOT_API_KEY` | `MOONSHOT_BASE_URL` | + +The openrouter, siliconflow, and custom groups speak the OpenAI-compatible protocol, hence the shared `OPENAI_*` variables. Provider groups and the built-in model catalog are covered in [Models & Providers](/models). + +## Project config + +`//.project_config.toml` is the Project's single config file: a hidden file written with mode 0600, with credentials inlined on the model entries. Model identity is always the `(provider, model_id)` pair — string concatenation is forbidden everywhere. + +| Field | Description | +| --- | --- | +| `name` | Project display name (the id is shown when unset) | +| `default_model` | Paired reference `{ provider, model_id }` to the default model; must point to an entry in `models` | +| `vision_model` | The vision model that reads images on behalf of text-only models (used by `describe_image`); a paired reference | +| `[[models]]` | The list of available model entries | + +Model entry (`[[models]]`) fields: + +| Field | Description | +| --- | --- | +| `provider` | Provider group; together with `model_id` forms the entry's unique key | +| `model_id` | Upstream request id, sent to AgentHub unchanged | +| `context_window` | Context window size | +| `client_type` | AgentHub client protocol; inferred from `model_id` by default — third-party OpenAI-compatible models should set `openai` | +| `display_name` | Display name; persisted only when it differs from the built-in catalog | +| `vision` | Whether image input is supported; defaults to supported | +| `pricing` | Three price buckets `cache_read` / `cache_write` / `output`, in USD per million Tokens (`unit = "usd_per_mtok"`) | +| `api_key` | Inline credential; when empty, falls back to the provider environment variable | +| `base_url` | Custom base URL; preset for gateway models | +| `created_at` | Write timestamp of `api_key` (ISO 8601; a display field maintained by the interface layer) | + +```toml +default_model = { provider = "deepseek", model_id = "deepseek-v4-pro" } + +[[models]] +provider = "deepseek" +model_id = "deepseek-v4-pro" +context_window = 1000000 +vision = false +api_key = "sk-..." + +[models.pricing] +unit = "usd_per_mtok" +cache_read = 0.003571 +cache_write = 0.428571 +output = 0.857143 +``` + +`pricing.unit` is currently always `usd_per_mtok` (USD per million tokens); the three buckets map onto `token_usage`'s three counters. + +Edit this file via the CLI (`penguin config model …`) or the Web Models page — never by hand while the service is running, and never by the model itself, which has no right to read or write it. + +## Agent config + +`agent_state/system_config.yaml` defines a single Agent's behavior (YAML; comments are preserved when edited via the Web UI): + +| Field | Default | Description | +| --- | --- | --- | +| `name` | — | Agent display name (falls back to the id) | +| `description` | — | Agent description | +| `version` | `1` | Agent State version (a natural number), incremented on each successful optimization | +| `system_prompt` | built-in template | Required; the only template with placeholder substitution | +| `max_turns` | `100` | Maximum LLM turns per Task | +| `model.max_tokens` | `32000` | Output Token limit per Request | +| `model.thinking_level` | `medium` | `none` / `low` / `medium` / `high` / `xhigh` | +| `model.timeoutMs` | `120000` | Per-Request timeout (milliseconds) | +| `compaction.max_context_length` | `128000` | Context Token threshold that triggers compaction | +| `compaction.max_session_turns` | `-1` | Cumulative Session turn threshold (`-1` = unlimited) | +| `compaction.mode` | `summarize` | `summarize` / `discard` | +| `compaction.prompt` | built-in template | Prompt used for summarize compaction | +| `tools.builtin` | full default toolset when omitted | Tool entries: `name` / `description` / `parameters` / `permission` (`r` or `rw`) / `forModel` / `timeoutMs` / `maxOutputLength`; once written it replaces the default list wholesale | +| `tools.mcpServers` | `[]` | MCP Server configuration (`name` + `config`); reserved for the MCP adapter layer | + +Tool permissions and approval semantics are covered in [Tools & Approval](/tools). + +A partial-override example (edit the file the init step generated). Note that this file is **not deep-merged with the defaults**: a key you write out takes effect wholesale, and only omitted keys fall back to the defaults above at their use sites; `system_prompt` is required (loading refuses without it), so keep the full generated template when editing other fields: + +```yaml +name: default_agent +description: General-purpose agent +version: 3 + +# Required: keep the full generated default template ({{AGENTS_MD}} and friends; elided here). +system_prompt: | + … + +max_turns: 100 + +model: + max_tokens: 32000 + thinking_level: medium + timeoutMs: 120000 + +compaction: + max_context_length: 128000 + max_session_turns: -1 + mode: summarize + +# Omitting the whole tools section = the full default toolset. Writing tools.builtin +# REPLACES the default list wholesale: carry the complete definition (including the +# parameters JSON Schema) for every tool you keep — see Tools & Approval. +``` + +### System prompt placeholders + +`system_prompt` is the only template with placeholder substitution. Available placeholders: + +| Placeholder | Injected content | +| --- | --- | +| `{{AGENTS_MD}}` | Full text of `AGENTS.md` | +| `{{VAULT_KEYS}}` | List of Vault key names (names only) | +| `{{SKILL_METADATA}}` | Metadata of installed Skills | +| `{{PLATFORM}}` | Runtime platform | +| `{{OS_VERSION}}` | Operating system version | +| `{{DATE}}` | Current date | +| `{{CWD}}` | Workspace path | +| `{{AGENT_ID}}` | Agent id | +| `{{PROJECT_DIR}}` | Project directory | +| `{{SESSION_ID}}` | Session id | + +`agent_state/AGENTS.md` is the developer-editable instruction file, injected via `{{AGENTS_MD}}` and empty by default — it is also the file an optimizer edits most (see [Self-Improvement](/self-improvement)). + +## Vault + +`agent_state/.vault.toml` is the Agent-level environment-variable vault: a hidden file written with mode 0600. + +- Key names must match `^[A-Za-z_][A-Za-z0-9_]*$` (shell environment variable naming rules); +- Values are injected only into tool subprocess environments and never enter the model context or the Trace; +- Only key names are disclosed in the system prompt via `{{VAULT_KEYS}}`; +- Managed via `penguin config vault set/list/remove` or the Web Vault tab. + +## Schedules + +Each file `agent_state/schedule/.toml` describes one scheduled task (the filename is its identity) that sends a preset Prompt to the Agent on a cadence. Schedules execute only while the Web service (the server runtime) is running, and are managed in the Web Agent settings → Schedule tab. + +| Field | Required | Description | +| --- | --- | --- | +| `prompt` | yes | The Prompt sent on each trigger | +| `enabled` | no | Enabled switch; defaults to `false` | +| `start_at` | yes | First trigger time (ISO 8601) | +| `period` | no | Cadence such as `30m` / `12h` / `7d`, minimum 5 minutes; omitted means a one-shot task | +| `end_at` | no | End time; must be later than `start_at` | +| `session_id` | no | Bind to an existing Session; mutually exclusive with the three fields below | +| `workspace` | no | Workspace for new-Session mode | +| `provider` / `model_id` | no | Paired model reference for new-Session mode | + +```toml +prompt = "Check yesterday's builds and summarize the failures" +enabled = true +start_at = 2026-08-01T09:00:00Z +period = "12h" +``` + +## Design principle + +An Agent's behavior lives entirely in editable files on disk — prompts, Skills, and configuration are data, not code. That is what makes Agents improvable by Agents: an optimizer edits exactly the same files you edit by hand. See [Self-Improvement](/self-improvement) and the [CLI Reference](/cli). diff --git a/packages/docs/content/configuration.zh.md b/packages/docs/content/configuration.zh.md new file mode 100644 index 0000000..f24e2d0 --- /dev/null +++ b/packages/docs/content/configuration.zh.md @@ -0,0 +1,186 @@ +--- +title: 配置参考 +description: 环境变量、Project 配置、Agent 配置、Vault 与定时任务的完整字段参考。 +--- + +PenguinHarness 的配置分三层:环境变量决定部署形态,Project 配置管理模型与凭证,Agent 配置定义单个 Agent 的行为。此外每个 Agent 还有 Vault(私有环境变量)与 Schedule(定时任务)两类状态文件。 + +## 环境变量 + +CLI 与服务端启动时会自动加载工作目录下的 `.env` 文件。 + +| 变量 | 说明 | 缺省值 | +| --- | --- | --- | +| `PENGUIN_HOME` | 数据根目录 | `~/.penguin/data` | +| `PORT` | Web 服务监听端口 | `7364` | +| `HOST` | Web 服务监听地址 | `127.0.0.1` | +| `PENGUIN_WEB_DB` | 服务端 SQLite 数据库路径 | `/web.db` | +| `PENGUIN_WEB_DIST` | 前端静态资源目录 | npm 安装的服务端包回退到内置 web-dist | +| `PENGUIN_LANG` | CLI 语言(`en` / `zh`),用 `penguin config lang` 设置 | `en` | + +### Provider 凭证环境变量 + +当模型条目未内联 `api_key` 时,AgentHub 网关按 Provider 回退读取对应环境变量;`*_BASE_URL` 变体同理覆盖 Base URL: + +| Provider | API Key | Base URL | +| --- | --- | --- | +| deepseek | `DEEPSEEK_API_KEY` | `DEEPSEEK_BASE_URL` | +| anthropic | `ANTHROPIC_API_KEY` | `ANTHROPIC_BASE_URL` | +| openai、openrouter、siliconflow、custom | `OPENAI_API_KEY` | `OPENAI_BASE_URL` | +| google | `GEMINI_API_KEY` | `GEMINI_BASE_URL` | +| zhipu | `ZAI_API_KEY` | `ZAI_BASE_URL` | +| moonshot | `MOONSHOT_API_KEY` | `MOONSHOT_BASE_URL` | + +openrouter、siliconflow 与 custom 分组走 OpenAI 兼容协议,因此复用 `OPENAI_*` 变量。Provider 分组与内置模型目录见[模型与 Provider](/models)。 + +## Project 配置 + +`//.project_config.toml` 是 Project 唯一的配置文件:隐藏文件,落盘权限 0600,凭证内联在模型条目上。模型身份始终是 `(provider, model_id)` 成对引用,禁止任何形式的字符串拼接。 + +| 字段 | 说明 | +| --- | --- | +| `name` | Project 展示名(缺省显示 id) | +| `default_model` | 缺省模型的成对引用 `{ provider, model_id }`,必须指向 `models` 中的条目 | +| `vision_model` | 代读图片的视觉模型(供纯文本模型的 `describe_image` 使用),成对引用 | +| `[[models]]` | 可用模型条目列表 | + +模型条目(`[[models]]`)字段: + +| 字段 | 说明 | +| --- | --- | +| `provider` | Provider 分组;与 `model_id` 共同构成条目唯一键 | +| `model_id` | 上游请求 id,原样发送给 AgentHub | +| `context_window` | 上下文窗口大小 | +| `client_type` | AgentHub 客户端协议;缺省由 `model_id` 推断,OpenAI 兼容的第三方模型应设为 `openai` | +| `display_name` | 展示名;仅在与内置目录不同时持久化 | +| `vision` | 是否支持图片输入;缺省视为支持 | +| `pricing` | 三档价格 `cache_read` / `cache_write` / `output`,单位 USD 每百万 Token(`unit = "usd_per_mtok"`) | +| `api_key` | 内联凭证;留空回退到 Provider 环境变量 | +| `base_url` | 自定义 Base URL;网关模型预置 | +| `created_at` | `api_key` 写入时间(ISO 8601,界面维护的展示字段) | + +```toml +default_model = { provider = "deepseek", model_id = "deepseek-v4-pro" } + +[[models]] +provider = "deepseek" +model_id = "deepseek-v4-pro" +context_window = 1000000 +vision = false +api_key = "sk-..." + +[models.pricing] +unit = "usd_per_mtok" +cache_read = 0.003571 +cache_write = 0.428571 +output = 0.857143 +``` + +`pricing.unit` 目前固定为 `usd_per_mtok`(USD 每百万 Token);三档对应 `token_usage` 的三个计数桶。 + +该文件通过 CLI `penguin config model …` 或 Web 的 Models 页面修改——服务运行期间不要手工编辑,模型本身则永远无权读写它。 + +## Agent 配置 + +`agent_state/system_config.yaml` 定义单个 Agent 的行为(YAML;经 Web UI 编辑时保留注释): + +| 字段 | 缺省值 | 说明 | +| --- | --- | --- | +| `name` | — | Agent 展示名(缺省回退到 id) | +| `description` | — | Agent 描述 | +| `version` | `1` | Agent State 版本号(自然数),每次成功优化自增 | +| `system_prompt` | 内置模板 | 必填;唯一进行占位符替换的模板 | +| `max_turns` | `100` | 单个 Task 的最大 LLM 轮数 | +| `model.max_tokens` | `32000` | 单次输出 Token 上限 | +| `model.thinking_level` | `medium` | `none` / `low` / `medium` / `high` / `xhigh` | +| `model.timeoutMs` | `120000` | 单次 Request 超时(毫秒) | +| `compaction.max_context_length` | `128000` | 触发压缩的上下文 Token 阈值 | +| `compaction.max_session_turns` | `-1` | Session 累计轮数阈值(`-1` 不限制) | +| `compaction.mode` | `summarize` | `summarize` / `discard` | +| `compaction.prompt` | 内置模板 | summarize 压缩使用的 Prompt | +| `tools.builtin` | 缺省时为完整默认工具集 | 工具条目:`name` / `description` / `parameters` / `permission`(`r` 或 `rw`)/ `forModel` / `timeoutMs` / `maxOutputLength`;一旦写出即整体替换默认列表 | +| `tools.mcpServers` | `[]` | MCP Server 配置(`name` + `config`),预留给 MCP 适配层 | + +工具权限与审批语义见[工具与审批](/tools)。 + +局部调整示例(在初始化生成的文件基础上修改)。注意本文件**不与默认值做 deep merge**:写出的字段整体生效,省略的字段才在使用处回退表中缺省值;`system_prompt` 是必填字段(缺失会拒绝加载),编辑其他字段时应保留初始化写入的完整模板: + +```yaml +name: default_agent +description: General-purpose agent +version: 3 + +# 必填:保留初始化生成的完整默认模板(含 {{AGENTS_MD}} 等占位符,此处从略)。 +system_prompt: | + … + +max_turns: 100 + +model: + max_tokens: 32000 + thinking_level: medium + timeoutMs: 120000 + +compaction: + max_context_length: 128000 + max_session_turns: -1 + mode: summarize + +# tools 整段省略 = 使用完整默认工具集。一旦写出 tools.builtin,将**整体替换** +# 默认列表:必须为每个要保留的工具携带完整定义(含 parameters JSON Schema), +# 参见「工具与审批」页。 +``` + +### 系统提示词占位符 + +`system_prompt` 是唯一进行占位符替换的模板,可用占位符: + +| 占位符 | 注入内容 | +| --- | --- | +| `{{AGENTS_MD}}` | `AGENTS.md` 的全文 | +| `{{VAULT_KEYS}}` | Vault 的键名列表(仅键名) | +| `{{SKILL_METADATA}}` | 已安装 Skill 的元数据 | +| `{{PLATFORM}}` | 运行平台 | +| `{{OS_VERSION}}` | 操作系统版本 | +| `{{DATE}}` | 当前日期 | +| `{{CWD}}` | Workspace 路径 | +| `{{AGENT_ID}}` | Agent id | +| `{{PROJECT_DIR}}` | Project 目录 | +| `{{SESSION_ID}}` | Session id | + +`agent_state/AGENTS.md` 是开发者可编辑的指令文件,经 `{{AGENTS_MD}}` 注入系统提示词,缺省为空——它也是优化器最常改动的文件(见[自我进化](/self-improvement))。 + +## Vault + +`agent_state/.vault.toml` 是 Agent 级的环境变量保险库:隐藏文件,落盘权限 0600。 + +- 键名须匹配 `^[A-Za-z_][A-Za-z0-9_]*$`(shell 环境变量命名规则); +- 值只注入工具子进程的环境变量,永远不进入模型上下文与 Trace; +- 系统提示词中经 `{{VAULT_KEYS}}` 只披露键名; +- 通过 CLI `penguin config vault set/list/remove` 或 Web 的 Vault 标签页管理。 + +## 定时任务 + +`agent_state/schedule/.toml` 每个文件描述一个定时任务(文件名即任务标识),按节律向 Agent 发送预设 Prompt。定时任务仅在 Web 服务(server 运行时)运行期间执行,在 Web 的 Agent 设置 → Schedule 标签页管理。 + +| 字段 | 必填 | 说明 | +| --- | --- | --- | +| `prompt` | 是 | 触发时发送的 Prompt | +| `enabled` | 否 | 是否启用,缺省 `false` | +| `start_at` | 是 | 首次触发时刻(ISO 8601) | +| `period` | 否 | 周期,形如 `30m` / `12h` / `7d`,下限 5 分钟;缺省为一次性任务 | +| `end_at` | 否 | 结束时刻,须晚于 `start_at` | +| `session_id` | 否 | 绑定既有 Session;与下列三项互斥 | +| `workspace` | 否 | 新建 Session 模式的 Workspace | +| `provider` / `model_id` | 否 | 新建 Session 模式的模型成对引用 | + +```toml +prompt = "检查昨日构建结果并汇总失败原因" +enabled = true +start_at = 2026-08-01T09:00:00Z +period = "12h" +``` + +## 设计原则 + +Agent 的行为完整地存放于磁盘上的可编辑文件——提示词、Skill、配置都是数据而非代码。正因如此,Agent 才能被 Agent 改进:优化器编辑的与你手工编辑的是同一批文件。参见[自我进化](/self-improvement)与 [CLI 参考](/cli)。 diff --git a/packages/docs/content/installation.en.md b/packages/docs/content/installation.en.md new file mode 100644 index 0000000..8f2bebe --- /dev/null +++ b/packages/docs/content/installation.en.md @@ -0,0 +1,79 @@ +--- +title: Installation +description: Install PenguinHarness via the install script, npm, or from source. +--- + +## Requirements + +- Linux / macOS (x64 or arm64): the install script ships platform tarballs with an official Node.js runtime bundled — no local Node needed. +- Other platforms, or installing via npm / from source: system Node.js >= 24. + +## Script install (recommended) + +On Linux / macOS: + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +``` + +The script downloads the matching `penguin-{linux,darwin}-{x64,arm64}.tar.gz`, which bundles an official Node.js runtime. Other platforms do **not** fall back automatically: the script exits and asks you to install Node.js >= 24 and re-run with `--universal`, which selects the runtime-less `penguin-universal.tar.gz`. + +Verify the install: + +```bash +penguin -v +``` + +### Install location and options + +| Item | Details | +| --- | --- | +| Install dir | `~/.penguin` by default; override with the `PENGUIN_INSTALL_DIR` env var | +| Command entry | A symlink `~/.local/bin/penguin` is created (the script warns if `~/.local/bin` is not on PATH) | +| Version pin | `PENGUIN_VERSION=vX.Y.Z` env var, or the `--version vX.Y.Z` script flag; defaults to the latest Release | +| Integrity check | Downloads are sha256-verified when the Release ships checksum assets | +| Upgrade | Re-run the install script; files are swapped atomically | + +Script flags are passed as `curl ... | sh -s -- --universal`. + +### Data directory + +The data directory defaults to `~/.penguin/data` — under the install home `~/.penguin`, but never modified by install or upgrade — and is overridable with the `PENGUIN_HOME` env var. Model configuration, Session records, and other data are preserved across upgrades. + +## npm install + +Requires system Node.js >= 24: + +```bash +npm install -g @prismshadow/penguin-cli +``` + +The npm package is `@prismshadow/penguin-cli`; the installed command is `penguin`. Web UI assets ship inside the `@prismshadow/penguin-server` package, so this single install yields a working `penguin web`. + +## From source + +Requires Node.js >= 24 and pnpm: + +```bash +git clone https://github.com/Prism-Shadow/penguin-harness.git +cd penguin-harness +pnpm install && pnpm build +``` + +After the build, run `pnpm penguin ` inside the repo as the dev runner, or use the globally linked `penguin` command. + +## Published npm packages + +| Package | Description | +| --- | --- | +| `@prismshadow/penguin-cli` | Command-line tool providing the `penguin` command | +| `@prismshadow/penguin-core` | SDK for creating Agents and Sessions programmatically | +| `@prismshadow/penguin-server` | Web service, including the Web UI assets | +| `@prismshadow/penguin-skills` | Skill collection | + +All packages are published under the Apache-2.0 license. + +## Next steps + +- [Quickstart](/quickstart): configure a model and run your first Task. +- [CLI Reference](/cli): the full list of commands and options. diff --git a/packages/docs/content/installation.zh.md b/packages/docs/content/installation.zh.md new file mode 100644 index 0000000..d7fc68b --- /dev/null +++ b/packages/docs/content/installation.zh.md @@ -0,0 +1,79 @@ +--- +title: 安装 +description: 通过安装脚本、npm 或源码安装 PenguinHarness。 +--- + +## 系统要求 + +- Linux / macOS(x64 或 arm64):安装脚本提供内置官方 Node.js 运行时的平台压缩包,解压即用,无需本机安装 Node。 +- 其他平台,或通过 npm / 源码安装:需要系统 Node.js >= 24。 + +## 脚本安装(推荐) + +在 Linux / macOS 上执行: + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +``` + +脚本按平台下载 `penguin-{linux,darwin}-{x64,arm64}.tar.gz`,其中捆绑了官方 Node.js 运行时。其他平台**不会自动回退**:脚本会退出并提示先安装 Node.js >= 24、再携带 `--universal` 重新执行,改用不含运行时的 `penguin-universal.tar.gz`。 + +安装完成后验证: + +```bash +penguin -v +``` + +### 安装位置与选项 + +| 项目 | 说明 | +| --- | --- | +| 安装目录 | 默认 `~/.penguin`,可用环境变量 `PENGUIN_INSTALL_DIR` 覆盖 | +| 命令入口 | 创建符号链接 `~/.local/bin/penguin`(若 `~/.local/bin` 不在 PATH 上,脚本会给出提示) | +| 版本固定 | 环境变量 `PENGUIN_VERSION=vX.Y.Z`,或脚本参数 `--version vX.Y.Z`;默认安装最新 Release | +| 完整性校验 | Release 提供 checksum 资产时自动进行 sha256 校验 | +| 升级 | 重新执行安装脚本即可,文件原子替换 | + +脚本参数通过 `curl ... | sh -s -- --universal` 的形式传入。 + +### 数据目录 + +数据目录默认位于 `~/.penguin/data`(在安装主目录 `~/.penguin` 之下,但安装与升级都不会改动它),可用环境变量 `PENGUIN_HOME` 覆盖。模型配置、Session 记录等在升级后均会保留。 + +## npm 安装 + +需要系统 Node.js >= 24: + +```bash +npm install -g @prismshadow/penguin-cli +``` + +npm 包名为 `@prismshadow/penguin-cli`,安装后的命令是 `penguin`。Web UI 静态资源随 `@prismshadow/penguin-server` 包发布,因此仅执行上述命令即可直接使用 `penguin web`。 + +## 源码安装 + +需要 Node.js >= 24 与 pnpm: + +```bash +git clone https://github.com/Prism-Shadow/penguin-harness.git +cd penguin-harness +pnpm install && pnpm build +``` + +构建完成后,在仓库内用 `pnpm penguin ` 作为开发入口运行,或使用全局链接的 `penguin` 命令。 + +## 已发布的 npm 包 + +| 包 | 说明 | +| --- | --- | +| `@prismshadow/penguin-cli` | 命令行工具,提供 `penguin` 命令 | +| `@prismshadow/penguin-core` | SDK,程序化创建 Agent 与 Session | +| `@prismshadow/penguin-server` | Web 服务,含 Web UI 静态资源 | +| `@prismshadow/penguin-skills` | Skill 集合 | + +全部包以 Apache-2.0 协议发布。 + +## 下一步 + +- [快速开始](/quickstart):配置模型并运行第一个 Task。 +- [CLI 参考](/cli):完整的命令与选项列表。 diff --git a/packages/docs/content/interfaces.en.md b/packages/docs/content/interfaces.en.md new file mode 100644 index 0000000..7ceb61c --- /dev/null +++ b/packages/docs/content/interfaces.en.md @@ -0,0 +1,241 @@ +--- +title: Core Interfaces +description: A top-down tour of the contracts — full LLMInterface and EnvironmentInterface signatures, inner types field by field, and every swappable seam. +--- + +The context_engine depends on three interfaces: Human, LLM and Environment. All protocol conversion happens inside the implementations — the engine sees only [OmniMessage](/omni-message). This page goes top-down: the two big interface signatures and the Human boundary first, then each interface's inner types layer by layer. All types are exported by `@prismshadow/penguin-core`; source: `packages/core/src/interfaces.ts`. + +## Overview + +```text + Human (a boundary, not a class) + session.run(newMessages, { approve, signal }) + │ ▲ + ▼ │ streamed OmniMessage + context_engine + │ │ + LLMInterface │ │ EnvironmentInterface + ▼ ▼ + GenerativeModel Environment + └─ AgentHub gateway └─ BuiltinTool registry (exec_command …) +``` + +| Interface | Contract | Built-in implementation | +| --- | --- | --- | +| Human | `session.run`'s inputs and streamed output | CLI, Server (SSE) | +| LLM | `LLMInterface.streamGenerate` | `GenerativeModel` (over AgentHub) | +| Environment | `EnvironmentInterface.executeTool` et al. | `Environment` + the builtin tool registry | + +Two iron rules run through every interface: **never throw into the engine** (errors converge into messages/returns carrying a `stop_reason`), and **the streaming discipline** (`start → delta → stop`, complete message immediately after). + +## LLMInterface + +The complete model-side contract is a single method: + +```ts +interface LLMInterface { + streamGenerate(parameters: GenerativeModelParameters): AsyncGenerator; +} + +interface GenerativeModelParameters { + newMessages: OmniMessage[]; // only this turn's new messages (the impl owns history; mixed roles rejected) + signal?: AbortSignal; +} +``` + +The generator yields `partial_*` fragments and complete messages, emits Token usage as `token_usage` events, and reports the terminal state via its **return value** (not a yielded message). + +### LLMOutcome semantics + +```ts +interface LLMOutcome { + status: StopReason; // completed | timeout | malformed | aborted | failed + message?: string; // display text when failed +} +``` + +| status | Meaning | Engine reaction | +| --- | --- | --- | +| `completed` | finished normally (token_usage already emitted) | proceed | +| `timeout` | timeout / lost connection | auto-reconnect within the run | +| `malformed` | response parse failure | auto-reconnect within the run | +| `aborted` | user interrupt | stop, hand back to the user | +| `failed` | non-retryable (auth/params, …) | stop, hand back to the user | + +Implementation constraints: never throw; no internal retries — reconnecting is the engine's job (see [The Agent Loop](/agent-loop)). + +### GenerativeModelConfig + +The built-in implementation's init config, field by field: + +```ts +interface GenerativeModelConfig { + modelId: string; + apiKey?: string; + baseUrl?: string; + clientType?: string; // AgentHub client protocol (openai / …); inferred from modelId when omitted + tools: ToolDefinition[]; + systemPrompt?: string; // fully assembled system prompt, placeholders substituted + contextWindow?: number; + maxTokens?: number; + thinkingLevel?: ThinkingLevelName; // "none" | "low" | "medium" | "high" | "xhigh" + requestTimeoutMs?: number; // per-Request timeout, default 120000; <=0 disables + toolCallIds?: ToolCallIdAllocator; // Session-level tool_call_id registry (pass the same instance across compaction) +} +``` + +### The built-in implementation: GenerativeModel + +`GenerativeModel` (`packages/core/src/llm/generative-model.ts`) grounds the contract on the `AutoLLMClient` of the `@prismshadow/agenthub` model gateway: + +- the gateway maintains conversation history **statefully**, receiving only new messages each turn; resuming a Session replays committed history through a one-time `setHistory`; +- an internal `EventTranslator` translates gateway stream events into `partial_*` fragments plus complete messages, preserving the `signature` / `phase` fidelity fields; complete messages settle in thinking → text → tool_call order; +- `ToolCallIdAllocator` disambiguates providers that use the function name as the call id (append `#n` inbound, strip outbound), scoped to the whole Session; +- provider differences (tool-call formats, reasoning content, streaming events) are absorbed entirely inside the gateway — see [Models & Providers](/models). + +## EnvironmentInterface + +The complete tool-execution contract: + +```ts +interface EnvironmentInterface { + listTools(): Promise; + executeTool(request: ToolExecutionRequest): AsyncGenerator; + toolPermission(name: string): "r" | "rw" | undefined; // for frontend approval-mode decisions + dispose?(): void; // release runtime resources; idempotent +} +``` + +`executeTool` yields `partial_tool_call_output` fragments and ends with exactly one complete `tool_call_output`; `origin`-tagged nested messages (e.g. forwarded by `run_subagent`) pass through unchanged. Rendering is explicitly not this interface's concern — streaming rendering belongs to the CLI / Web front ends. + +### ToolExecutionRequest and EnvironmentConfig + +```ts +interface ToolExecutionRequest { + toolCall: OmniMessage; // an approved call + signal?: AbortSignal; + approve?: ApproveFn; // forwarded to tools that spawn child Sessions (approval inheritance) +} + +interface EnvironmentConfig { + workspaceDir: string; + toolConfig: ToolConfig; // { customTools: ToolDefinitionConfig[]; mcpServers: MCPServerConfig[] } + services?: EnvironmentServices; // runtime services injected into individual tools + vault?: Record; // Vault env vars, injected into exec_command / input_command subprocesses +} + +interface EnvironmentServices { + subagentRunner?: SubagentRunner; // needed by run_subagent + visionDescriber?: VisionDescriberService; // needed by describe_image on text-only models + commandSessions?: CommandSessionManager; // long-running command session registry (built by Environment) + subagentSessions?: SubagentSessionManager;// background subagent session registry (likewise) +} + +interface MCPServerConfig { + name: string; + config: Record; +} +``` + +### The inner tool contract: BuiltinTool + +Inside the Environment, an individual tool follows a deliberately narrower contract ("loose tool, strict framework"): + +```ts +interface BuiltinTool { + name: string; + definition: ToolDefinitionConfig; + execute( + args: Record, + ctx: ToolExecutionContext, // { workspaceDir, toolCallId, signal?, approve? } + ): AsyncGenerator; +} + +interface ToolDefinitionConfig { + name: string; + description: string; + parameters?: Record; // JSON Schema + permission?: "r" | "rw"; + forModel?: "vision" | "text-only"; // assembled per session-model class + timeoutMs?: number; // default 120000; <=0 disables + maxOutputLength?: number; // default 16000, head-kept truncation; <=0 disables +} +``` + +A tool emits only content deltas; framing, timeouts, truncation, `stop_reason` priority and errors-to-messages are all handled centrally by the Environment — it is close to impossible for a tool author to break the protocol. Extension is registration: add one `name → factory` entry to `BUILTIN_TOOL_FACTORIES` (`packages/core/src/environment/tools/registry.ts`). Per-tool parameters and behavior: [Tools & Approval](/tools). + +## The Human boundary + +Human is deliberately not an interface class. The SDK caller *is* the Human: + +```ts +const session = await agent.createSession({ workspaceDir, modelId }); + +session.run( + newMessages: OmniMessage[], // input: the Prompt + opts?: RunOptions, +): AsyncGenerator; // output: streamed OmniMessage + +interface RunOptions { + signal?: AbortSignal; // interrupt (e.g. Ctrl-C) + approve?: ApproveFn; // per-tool approval; denies everything when omitted +} +``` + +The CLI wires terminal I/O onto this boundary; the Server wires HTTP requests and SSE channels onto it. Any programmatic caller that connects becomes a new Human implementation — nothing to register. + +## ApproveFn + +```ts +type ApprovalDecision = "allow" | "deny"; +type ApproveFn = (toolCall: OmniMessage) => Promise; +``` + +Constraints: called exactly once per complete `tool_call`; a throwing callback counts as `deny`; when none is injected the engine denies everything (conservative default). A Subagent inherits its parent's approval callback (invoked with an `origin` tag), so the approval policy spans the whole delegation tree. + +## Subagent interfaces + +Subagent creation is injected at the `createAgent` composition layer, so the Environment never back-depends on the layers above it: + +```ts +interface SubagentRunner { + // Precheck errors (depth limit, unknown agent) are thrown — Environment collapses them to failed + spawn(input: { + agentId?: string; // defaults to the current Agent (self-spawn) + modelId?: string; // defaults to the Project default model + }): Promise; +} + +interface SubagentHandle { + sessionId: string; // the child Session id: the origin hop; subagent_id derives from its tail + run(input: { + prompt: string; + signal?: AbortSignal; + approve?: ApproveFn; // the parent's approval callback — forwarding is inheritance + }): AsyncGenerator; + dispose(): void; // release the child Session's runtime resources; idempotent +} +``` + +Spawning and running are separate, so the same child Session can accept a follow-up Prompt after a turn ends (a long-running Subagent, driven via `input_subagent`). Child Sessions run in the same Workspace with their own Trace; nesting depth is currently capped at 1. + +## VisionDescriberService + +The image proxy-reading service for text-only models (needed by `describe_image`): + +```ts +interface VisionDescriberService { + modelId: string | null; // null when the Project has no vision_model — the tool ends with a failed explanation + createLLM?: () => LLMInterface; // one-shot LLM for the vision model (no tools, no system prompt) +} +``` + +## Extension seams + +| To … | Do … | +| --- | --- | +| Swap or customize model access | implement `LLMInterface` (or just set `client_type` for OpenAI-compatible endpoints) | +| Swap the execution sandbox | implement `EnvironmentInterface` | +| Add a tool | implement `BuiltinTool` + register a factory; or declare it under `tools.builtin` in `system_config.yaml` | +| Customize approval policy | inject an `ApproveFn` (the CLI/Web modes are wrappers over it) | +| Change an Agent's behavior | edit its Agent State: `system_config.yaml`, `AGENTS.md`, Skills — see the [Configuration Reference](/configuration) | diff --git a/packages/docs/content/interfaces.zh.md b/packages/docs/content/interfaces.zh.md new file mode 100644 index 0000000..03fab56 --- /dev/null +++ b/packages/docs/content/interfaces.zh.md @@ -0,0 +1,241 @@ +--- +title: 接口契约 +description: 自顶向下的接口全览:LLMInterface 与 EnvironmentInterface 的完整签名、内层类型逐字段定义,以及每一处可替换的扩展点。 +--- + +context_engine 依赖三个接口:Human、LLM、Environment。协议转换全部发生在接口实现内部——引擎只见 [OmniMessage](/omni-message)。本页自顶向下:先给出两大接口的完整签名与 Human 边界,再逐层展开每个接口的内部类型。类型全部由 `@prismshadow/penguin-core` 导出,源码见 `packages/core/src/interfaces.ts`。 + +## 总览 + +```text + Human(边界,非接口类) + session.run(newMessages, { approve, signal }) + │ ▲ + ▼ │ 流式 OmniMessage + context_engine + │ │ + LLMInterface │ │ EnvironmentInterface + ▼ ▼ + GenerativeModel Environment + └─ AgentHub 网关 └─ BuiltinTool 注册表(exec_command …) +``` + +| 接口 | 契约 | 内置实现 | +| --- | --- | --- | +| Human | `session.run` 的入参与流式出参 | CLI、Server(SSE) | +| LLM | `LLMInterface.streamGenerate` | `GenerativeModel`(基于 AgentHub) | +| Environment | `EnvironmentInterface.executeTool` 等 | `Environment` + 内置工具注册表 | + +两条铁律贯穿所有接口:**从不向引擎抛异常**(错误收敛为带 `stop_reason` 的消息/返回值),**流式纪律**(`start → delta → stop`,随后立即产出完整消息)。 + +## LLMInterface + +模型侧的完整契约只有一个方法: + +```ts +interface LLMInterface { + streamGenerate(parameters: GenerativeModelParameters): AsyncGenerator; +} + +interface GenerativeModelParameters { + newMessages: OmniMessage[]; // 仅本轮新增消息(实现自行维护历史,多 role 不接受) + signal?: AbortSignal; +} +``` + +生成器逐条产出 `partial_*` 分片与完整消息,Token 用量以 `token_usage` 事件产出;终态经**返回值**(而非产出消息)给出。 + +### LLMOutcome 语义 + +```ts +interface LLMOutcome { + status: StopReason; // completed | timeout | malformed | aborted | failed + message?: string; // failed 时的展示文案 +} +``` + +| status | 含义 | 引擎的反应 | +| --- | --- | --- | +| `completed` | 正常完成(已产出 token_usage) | 继续下一步 | +| `timeout` | 超时/断连 | 同一 run 内自动重连 | +| `malformed` | 响应解析失败 | 同一 run 内自动重连 | +| `aborted` | 用户中断 | 停止交还用户 | +| `failed` | 鉴权/参数等不可重试错误 | 停止交还用户 | + +实现约束:从不抛异常;不做内部重试(重连是引擎的职责,见 [Agent 运行循环](/agent-loop))。 + +### GenerativeModelConfig + +内置实现的初始化配置,逐字段: + +```ts +interface GenerativeModelConfig { + modelId: string; + apiKey?: string; + baseUrl?: string; + clientType?: string; // AgentHub 客户端协议(openai / …);缺省按 modelId 推断 + tools: ToolDefinition[]; + systemPrompt?: string; // 占位符替换完成后的完整系统提示词 + contextWindow?: number; + maxTokens?: number; + thinkingLevel?: ThinkingLevelName; // "none" | "low" | "medium" | "high" | "xhigh" + requestTimeoutMs?: number; // 单次 Request 超时,默认 120000;<=0 关闭 + toolCallIds?: ToolCallIdAllocator; // Session 级 tool_call_id 唯一性登记表(压缩重建时传同一实例) +} +``` + +### 内置实现:GenerativeModel + +`GenerativeModel`(`packages/core/src/llm/generative-model.ts`)把契约落到模型网关 `@prismshadow/agenthub` 的 `AutoLLMClient` 上: + +- 网关**有状态**地维护会话历史,每轮只接收新消息;恢复 Session 时经一次性的 `setHistory` 重放已提交历史; +- 内部的 `EventTranslator` 把网关流式事件翻译为 `partial_*` 分片 + 完整消息,保留 `signature` / `phase` 保真字段,完整消息按 thinking → text → tool_call 顺序落盘; +- `ToolCallIdAllocator` 处理个别 Provider 用函数名充当调用 id 的情况(入站追加 `#n`、出站剥离),作用域覆盖整个 Session; +- Provider 协议差异(工具调用格式、思考内容、流式事件)全部在网关内抹平,见[模型与 Provider](/models)。 + +## EnvironmentInterface + +工具执行侧的完整契约: + +```ts +interface EnvironmentInterface { + listTools(): Promise; + executeTool(request: ToolExecutionRequest): AsyncGenerator; + toolPermission(name: string): "r" | "rw" | undefined; // 供前端审批模式判定 + dispose?(): void; // 释放运行时资源,幂等 +} +``` + +`executeTool` 逐条产出 `partial_tool_call_output`,并以恰好一条完整 `tool_call_output` 收尾;带 `origin` 的嵌套消息(如 `run_subagent` 转发的子 Session 消息)原样透传。渲染不是本接口的职责——流式渲染由 CLI / Web 前端完成。 + +### ToolExecutionRequest 与 EnvironmentConfig + +```ts +interface ToolExecutionRequest { + toolCall: OmniMessage; // 已通过审批的调用 + signal?: AbortSignal; + approve?: ApproveFn; // 转发给需要派生子 Session 的工具,实现审批继承 +} + +interface EnvironmentConfig { + workspaceDir: string; + toolConfig: ToolConfig; // { customTools: ToolDefinitionConfig[]; mcpServers: MCPServerConfig[] } + services?: EnvironmentServices; // 注入给个别工具的运行时服务 + vault?: Record; // Vault 环境变量,注入 exec_command / input_command 子进程 +} + +interface EnvironmentServices { + subagentRunner?: SubagentRunner; // run_subagent 所需 + visionDescriber?: VisionDescriberService; // text-only 模型的 describe_image 所需 + commandSessions?: CommandSessionManager; // 长驻命令会话登记表(Environment 内部构造) + subagentSessions?: SubagentSessionManager;// 后台 Subagent 会话登记表(同上) +} + +interface MCPServerConfig { + name: string; + config: Record; +} +``` + +### 内层工具契约:BuiltinTool + +Environment 之内,单个工具遵循更窄的契约(「松工具、紧框架」): + +```ts +interface BuiltinTool { + name: string; + definition: ToolDefinitionConfig; + execute( + args: Record, + ctx: ToolExecutionContext, // { workspaceDir, toolCallId, signal?, approve? } + ): AsyncGenerator; +} + +interface ToolDefinitionConfig { + name: string; + description: string; + parameters?: Record; // JSON Schema + permission?: "r" | "rw"; + forModel?: "vision" | "text-only"; // 按 Session 模型类别装配 + timeoutMs?: number; // 默认 120000;<=0 关闭 + maxOutputLength?: number; // 默认 16000,头部保留截断;<=0 关闭 +} +``` + +工具只产出内容增量;封帧、超时、截断、`stop_reason` 优先级、错误转消息全部由 Environment 统一处理——工具作者几乎不可能写出破坏协议的工具。注册即扩展:向 `BUILTIN_TOOL_FACTORIES`(`packages/core/src/environment/tools/registry.ts`)添加一个 `名称 → 工厂` 条目即可。逐工具的参数与行为见[工具与审批](/tools)。 + +## Human 边界 + +Human 刻意不设计为接口类。SDK 的调用方就是 Human: + +```ts +const session = await agent.createSession({ workspaceDir, modelId }); + +session.run( + newMessages: OmniMessage[], // 输入:Prompt + opts?: RunOptions, +): AsyncGenerator; // 输出:流式 OmniMessage + +interface RunOptions { + signal?: AbortSignal; // 中断信号(如 Ctrl-C) + approve?: ApproveFn; // 逐工具审批;未注入时默认全部拒绝 +} +``` + +CLI 把终端输入输出接到这个边界上;Server 把 HTTP 请求与 SSE 通道接上来。任何程序化调用方接上来就是一种新的 Human 实现,无需注册。 + +## ApproveFn + +```ts +type ApprovalDecision = "allow" | "deny"; +type ApproveFn = (toolCall: OmniMessage) => Promise; +``` + +约束:每个完整 `tool_call` 恰好被调用一次;回调抛出异常按 `deny` 处理;未注入时引擎默认全部拒绝(保守策略)。Subagent 继承父级的审批回调(调用时带 `origin` 标记),审批策略天然贯穿整个委托树。 + +## Subagent 接口 + +Subagent 的创建能力在 `createAgent` 组装层注入,避免 Environment 反向依赖上层: + +```ts +interface SubagentRunner { + // 深度超限、目标 Agent 不存在等前置错误以抛出表达(由 Environment 收敛为 failed) + spawn(input: { + agentId?: string; // 缺省复用当前 Agent(自派生) + modelId?: string; // 缺省用 Project 默认模型 + }): Promise; +} + +interface SubagentHandle { + sessionId: string; // 子 Session id:消息 origin 的一跳,subagent_id 由其尾部派生 + run(input: { + prompt: string; + signal?: AbortSignal; + approve?: ApproveFn; // 父级审批回调,转发即继承 + }): AsyncGenerator; + dispose(): void; // 释放子 Session 运行时资源,幂等 +} +``` + +派生(spawn)与运行(run)分离,同一子 Session 可以在一轮结束后接受追加 Prompt 继续运行(长驻 Subagent,经 `input_subagent` 驱动)。子 Session 在同一 Workspace 中运行、拥有独立 Trace;嵌套深度当前限制为 1。 + +## VisionDescriberService + +text-only 模型的图像代读服务(`describe_image` 所需): + +```ts +interface VisionDescriberService { + modelId: string | null; // Project 未配置 vision_model 时为 null,工具以 failed 说明收尾 + createLLM?: () => LLMInterface; // 构造该视觉模型的一次性 LLM(无工具、无系统提示词) +} +``` + +## 扩展点一览 + +| 想要 | 做法 | +| --- | --- | +| 更换/自定义模型接入 | 实现 `LLMInterface`(或仅配置 `client_type` 走 OpenAI 兼容协议) | +| 更换执行沙箱 | 实现 `EnvironmentInterface` | +| 新增工具 | 实现 `BuiltinTool` + 注册工厂;或在 `system_config.yaml` 的 `tools.builtin` 中声明 | +| 定制审批策略 | 注入 `ApproveFn`(CLI/Web 的四种模式即其封装) | +| 改变 Agent 行为 | 编辑 Agent State:`system_config.yaml`、`AGENTS.md`、Skills,见[配置参考](/configuration) | diff --git a/packages/docs/content/introduction.en.md b/packages/docs/content/introduction.en.md new file mode 100644 index 0000000..42cbe46 --- /dev/null +++ b/packages/docs/content/introduction.en.md @@ -0,0 +1,49 @@ +--- +title: Introduction +description: What PenguinHarness is, what ships in the box, and the design tenets behind it. +--- + +PenguinHarness is an open-source AI Agent harness — a complete TypeScript stack built for constructing and evolving agents. It deploys fully locally (your data never leaves the machine), runs on as little as a single CPU, and reaches 1000+ online and local models through one unified model gateway. + +In one line: **Efficient Self-Improving Harness for Everyone.** + +## The three pillars + +PenguinHarness is organized around three radiating concepts — the message protocol, the SDK, and the skill library — each carrying one pillar: + +| Pillar | Meaning | +| --- | --- | +| **Simplest Is the Best** | A deliberately minimal toolset over clean low-level interfaces: fewer tool calls, fewer Tokens, complex tasks done efficiently. | +| **Harness for Building Agents** | With the PenguinHarness SDK, an Agent builds complete Agent applications for you — autonomously, from scratch. | +| **Harness for Recursive Self-Improvement** | With PenguinHarness Skills, an Agent evaluates and optimizes itself, improving recursively over time. | + +## What ships in the box + +One install gives you four layers that share a single data directory and a single message protocol: + +| Component | Package | Description | +| --- | --- | --- | +| SDK | `@prismshadow/penguin-core` | The core engine: ReAct loop, the [OmniMessage protocol](/omni-message), the LLM and Environment [interface contracts](/interfaces), Agent State and Trace. | +| CLI | `@prismshadow/penguin-cli` | The `penguin` command: interactive REPL, one-shot task runs, model and Vault configuration. | +| Server | `@prismshadow/penguin-server` | The Web backend: HTTP [API and SSE streaming](/server-api), multi-user auth, Project authorization, usage statistics. | +| Web App | `@prismshadow/penguin-web` | The browser UI: multi-session chat, Agent management, skill library, model configuration, Trace observability and the evaluation center. | + +## Design tenets + +These principles run through every component; the design pages keep coming back to them: + +- **A minimal toolset**: the shell is the universal interface — file reads, writes and edits all go through `exec_command`. See [Tools & Approval](/tools). +- **Agents are editable data**: prompts, Skills and config are editable files on disk, not hardcoded constants — what you can see, an Agent can improve. See the [Configuration Reference](/configuration). +- **Everything observable**: every request, tool call and approval decision is appended to the [Trace](/sessions-and-traces); a Session restores fully from it. +- **Errors converge into messages**: model and tool failures never throw — they become messages the model can react to. See [The Agent Loop](/agent-loop). +- **Streaming first**: text streams token by token; tool calls and results appear live. +- **Model ↔ Agent decoupling**: an Agent never binds to a model; you pick one per Session. See [Models & Providers](/models). + +## A note on naming + +The unified message protocol is called **OmniMessage** in technical writing (marketing materials also call it Penguin Message). This documentation uses OmniMessage throughout. + +## Next steps + +- [Install](/installation) PenguinHarness, then run your first Task with the [Quickstart](/quickstart). +- Start the design docs at the [Architecture](/architecture) overview to see how the pieces fit together. diff --git a/packages/docs/content/introduction.zh.md b/packages/docs/content/introduction.zh.md new file mode 100644 index 0000000..e88c746 --- /dev/null +++ b/packages/docs/content/introduction.zh.md @@ -0,0 +1,49 @@ +--- +title: 产品介绍 +description: PenguinHarness 是什么,它由哪些部分组成,以及它的设计信条。 +--- + +PenguinHarness 是一个开源的 AI Agent Harness——为「构建 Agent」与「进化 Agent」而生的一整套 TypeScript 基础设施。它完全本地部署,数据不出机器,最低一颗 CPU 即可运行;通过统一的模型网关可接入 1000+ 在线与本地模型。 + +一句话概括:**Efficient Self-Improving Harness for Everyone.** + +## 三大支柱 + +PenguinHarness 的能力围绕三个递进的概念展开——消息协议、SDK、技能库,分别支撑三个支柱: + +| 支柱 | 含义 | +| --- | --- | +| **Simplest Is the Best** | 在干净的底层接口之上刻意保持极简的工具集:更少的工具调用、更少的 Token,高效完成复杂任务。 | +| **Harness for Building Agents** | 基于 PenguinHarness SDK,由一个 Agent 从零开始为你自主构建完整的 Agent 应用。 | +| **Harness for Recursive Self-Improvement** | 基于 PenguinHarness Skills,Agent 评估并优化自己,随时间递归进化。 | + +## 产品组成 + +一次安装即获得完整的四层交付物,它们共享同一套数据目录与同一个消息协议: + +| 组件 | 包名 | 说明 | +| --- | --- | --- | +| SDK | `@prismshadow/penguin-core` | 核心引擎:ReAct 循环、[OmniMessage 协议](/omni-message)、LLM 与 Environment [接口契约](/interfaces)、Agent State 与 Trace。 | +| CLI | `@prismshadow/penguin-cli` | 命令行 `penguin`:交互式 REPL、单次任务运行、模型与 Vault 配置。 | +| Server | `@prismshadow/penguin-server` | Web 服务端:HTTP [API 与 SSE 流式通道](/server-api)、多用户认证、Project 授权、用量统计。 | +| Web App | `@prismshadow/penguin-web` | 浏览器界面:多 Session 对话、Agent 管理、技能库、模型配置、Trace 观测与评估中心。 | + +## 设计信条 + +这些原则贯穿所有组件,后续每一页设计文档都会反复引用: + +- **极简工具集**:shell 是通用接口,文件读写与命令执行统一经 `exec_command` 完成,见[工具与审批](/tools)。 +- **Agent 是可编辑的数据**:Prompt、Skill、配置都是磁盘上的可编辑文件,而非硬编码——你能看到的,Agent 就能改进,见[配置参考](/configuration)。 +- **全量可观测**:每一次请求、工具调用与审批决策都以追加方式写入 [Trace](/sessions-and-traces),Session 可从 Trace 完整恢复。 +- **错误收敛为消息**:模型与工具的错误不抛异常,而是变成模型可以继续处理的消息,见 [Agent 运行循环](/agent-loop)。 +- **流式优先**:文本逐 Token 流出,工具调用与结果实时可见。 +- **模型与 Agent 解耦**:Agent 不绑定模型,每个 Session 创建时自由选择,见[模型与 Provider](/models)。 + +## 命名说明 + +统一消息协议在技术文档中称为 **OmniMessage**(产品宣传中也叫 Penguin Message)。本文档站一律使用 OmniMessage。 + +## 下一步 + +- [安装](/installation) PenguinHarness,然后跟随[快速开始](/quickstart)跑通第一个 Task。 +- 从[架构总览](/architecture)进入设计文档,理解各组件如何协作。 diff --git a/packages/docs/content/message-flow.en.md b/packages/docs/content/message-flow.en.md new file mode 100644 index 0000000..9162261 --- /dev/null +++ b/packages/docs/content/message-flow.en.md @@ -0,0 +1,117 @@ +--- +title: Message Flow & Ordering +description: How messages travel between Human, engine, LLM, Environment and Trace — every ordering guarantee and non-guarantee, and why stream order differs from context order. +--- + +[The OmniMessage Protocol](/omni-message) defines what messages *are*; this page explains how they *move* and in what order they become visible: the delivery paths, the merge mechanism, the observable timeline within a turn, which orderings are guaranteed, which are not, and why "order on the stream" and "order in the model context" are two different things. Source of truth: `packages/core/src/engine/context-engine.ts`. + +## Delivery paths within a turn + +Five actors: Human (the SDK caller), engine (context_engine), LLM, Environment, Trace. Within one turn: + +```text +Human ──run(newMessages)──► engine + engine ──write Prompt──────────────────► Trace + engine ──request_begin──► Human and Trace + engine ──streamGenerate(new messages)──► LLM + ┌──────────── LLM streams partial_* and complete messages ────────┐ + │ engine forwards each: simultaneously ──► Human (yield) │ + │ and ──► Trace (write) │ + └──────────────────────────────────────────────────────────────────┘ + complete tool_call ──► engine: await approve(tc) (one at a time) + engine ──approval_decision──► Human and Trace + allow ──► Environment.executeTool (concurrent, never blocks the LLM stream) + Environment ──partial_tool_call_output──► Human, and (complete) ──► Trace + LLM stream ends: token_usage is its last message, request_end follows at once + still-running tools keep streaming output (possibly after request_end) + all outputs settled ──► reordered to original call order as the next turn's LLM input +``` + +Key point: **every message is written to the Trace at the same moment it enters the output stream**, so stream order and Trace order agree (the Trace merely skips partials and `origin`-tagged messages — see [Sessions & Traces](/sessions-and-traces)). + +## The merge point: MergeQueue + +A turn has several concurrent producers: the driver task consuming the LLM stream, plus N concurrently executing tools. All of them push into one merge queue, and a **single consumer** (the `run` generator) yields messages one at a time in **arrival order**; the turn ends only when every producer has finished and the queue is drained. + +This one mechanism fixes three basic properties of message delivery: + +1. the consumer sees a single **totally ordered** stream — no client-side multiplexing needed; +2. messages from different producers interleave by arrival time — tool outputs arrive in **completion order**, unrelated to call order; +3. order *within* one producer is preserved (the LLM stream is internally ordered; a single tool's fragments are ordered). + +## The observable order within a turn + +A turn with two tool calls, as the consumer observes it (annotated): + +```text + 1 event request_begin + 2 partial partial_thinking(start → delta… → stop) + 3 complete thinking ← the complete message right after stop + 4 partial partial_text(start → delta… → stop) + 5 complete text + 6 partial partial_tool_call A(start → delta… → stop) + 7 complete tool_call A + 8 event approval_decision(allow, A) ← approvals are sequential; A starts executing + 9 partial partial_tool_call B(…) ← the LLM stream continues, not waiting for A +10 complete tool_call B +11 event approval_decision(allow, B) +12 partial partial_tool_call_output B(…) ← B produces output first: completion order +13 complete tool_call_output B +14 event token_usage ← the LLM stream's last message +15 event request_end(completed) ← emitted when the LLM stream ends, not waiting for tools +16 partial partial_tool_call_output A(…) ← late output lands after request_end +17 complete tool_call_output A + (A and B settled → re-fed in A, B original order → next request_begin) +``` + +If a `tool_call` is denied, line 8 carries `deny` and a synthetic `aborted` `tool_call_output` ("Tool call denied by user.") follows immediately — nothing is dispatched. + +## Guarantees and non-guarantees + +**Guaranteed:** + +| Guarantee | Meaning | +| --- | --- | +| Streaming discipline | every segment goes strictly `start → delta* → stop`, complete message right after; concatenated deltas ≡ the complete message | +| Approval position | `approval_decision` comes after its `tool_call` and before any output of that tool | +| Pairing | every committed `tool_call` gets exactly one complete `tool_call_output` (a denial gets the synthetic one) | +| LLM stream tail | `token_usage` is the LLM stream's last message, `request_end` follows immediately | +| Commit criterion | `request_end.status === "completed"` ⇔ the turn was committed by the gateway (replay keeps or drops on this) | +| Stream order = Trace order | written as streamed; the Trace only filters partials and `origin` messages | +| Transport ordering | SSE delivers per channel with monotonic ids; reconnects replay from `Last-Event-ID` or get `resync_required` — see [Server API](/server-api) | + +**Not guaranteed (renderers must not rely on these):** + +| Non-guarantee | Meaning | +| --- | --- | +| Tool-output order | arrival is completion order; fragments of different tools interleave — attribute by `tool_call_id` | +| `request_end` ≠ end of turn | still-running tools may emit output after `request_end` and before the next `request_begin` | +| Event/content spacing | later LLM-stream messages may land between an `approval_decision` and that tool's first output | + +## Stream order vs context order + +The same batch of tool outputs exists in two orders, serving two different consumers: + +- **stream order (completion order)** — for the Human: whoever finishes first is visible first, for real-time rendering; +- **context order (original call order)** — for the model: before entering the next turn's input, outputs are reordered to the original `tool_call` order, matching provider pairing rules. + +Therefore **a renderer must never reconstruct the context from arrival order** — hang each output onto its call via `tool_call_id`; the engine owns context ordering. + +## Edge-case timelines + +| Case | Observable order on the stream | +| --- | --- | +| User interrupt | (messages produced so far) → the `abort` event — the last message before `run` returns; carry-over goes to the model context only, never streamed, never written to Trace | +| Automatic reconnect | `request_end(timeout \| malformed)` → a fresh `request_begin`; the `` block is model-visible only | +| Compaction | `compaction_begin` → the compaction request runs against the old context (its streamed output is **not** forwarded, only written to Trace) → that request's `token_usage` → `compaction_end(status)` | +| max_turns reached | a length notice → the run ends; unsubmitted input is kept as carry-over | +| The Prompt itself | written to Trace, not echoed back onto the stream (the caller already has it) | +| session_meta | never emitted on the main Session's stream (it lives in the Trace and the history API); a Subagent child stream's **first** message is the child's `session_meta` | + +## Across Sessions: the origin chain + +A child Session spawned by `run_subagent` has its own complete stream. When forwarded to the parent, each child message gets one child-Session-id hop prepended to `origin`, and it interleaves with the parent's own messages **by arrival time**; renderers route by `origin` into the nested card. Child messages are not written to the parent Trace — the parent keeps only the `subagent` pointer event, while the child's stream order is recorded in its own Trace. + +## Transport ordering (SSE) + +The Server pushes this exact output stream verbatim (single-line JSON) onto the per-Session SSE channel: monotonically increasing event ids, a bounded replay buffer for reconnects, `resync_required` when the replay window is gone. Event order: on reconnect the replayed gap (or `resync_required`) comes first, then the authoritative `task_state` snapshot and pending approvals; a fresh connection skips replay, so `task_state` is its first event. Details — including the bundled Web App's connect-first + dedup consumption pattern — are on the [Server API](/server-api) page. diff --git a/packages/docs/content/message-flow.zh.md b/packages/docs/content/message-flow.zh.md new file mode 100644 index 0000000..508ecc8 --- /dev/null +++ b/packages/docs/content/message-flow.zh.md @@ -0,0 +1,116 @@ +--- +title: 消息流转与时序 +description: 消息在 Human、engine、LLM、Environment 与 Trace 之间的传递机制,每一处顺序保证与非保证,以及流序与上下文序的区别。 +--- + +[OmniMessage 协议](/omni-message)定义了消息**是什么**,本页讲清消息**怎么传、以什么顺序可见**:传递路径、汇流机制、一轮内的可见时序、哪些顺序有保证、哪些没有,以及"流上的顺序"与"模型上下文的顺序"为何是两回事。源码依据:`packages/core/src/engine/context-engine.ts`。 + +## 一轮的传递路径 + +五个参与者:Human(SDK 调用方)、engine(context_engine)、LLM、Environment、Trace。一轮之内: + +```text +Human ──run(newMessages)──► engine + engine ──写 Prompt──────────────────────► Trace + engine ──request_begin──► Human 与 Trace + engine ──streamGenerate(新消息)──► LLM + ┌───────────────── LLM 流式返回 partial_* 与完整消息 ────────┐ + │ engine 逐条转发:每条同时 ──► Human(yield)与 ──► Trace(写) │ + └─────────────────────────────────────────────────────────────┘ + 完整 tool_call ──► engine:await approve(tc)(逐个) + engine ──approval_decision──► Human 与 Trace + allow ──► Environment.executeTool(并发,不阻塞 LLM 流) + Environment ──partial_tool_call_output──► Human 与(完整时)Trace + LLM 流结束:最后一条 token_usage,随即 request_end ──► Human 与 Trace + 仍在执行的工具继续流出输出(可晚于 request_end) + 全部输出齐 ──► 按原始调用顺序重排,作为下一轮 LLM 输入 +``` + +要点:**每条消息在进入输出流的同时写入 Trace**,因此流序与 Trace 序一致(Trace 跳过分片与带 `origin` 的消息,见 [Session 与 Trace](/sessions-and-traces))。 + +## 单一汇流点:MergeQueue + +一轮内存在多个并发生产者:消费 LLM 流的驱动任务,加上 N 个并发执行的工具。它们全部 push 进同一个合并队列,由**单一消费者**(`run` 生成器)按**到达顺序**逐条 yield;生产者全部完成且队列排空,这一轮才结束。 + +这一机制决定了消息传递的三条基本性质: + +1. 消费方看到的是一条**全序**的消息流,不需要自己做多路归并; +2. 不同生产者的消息按到达时刻交错——工具输出的先后是**完成顺序**,与调用顺序无关; +3. 同一生产者内部的顺序被保留(LLM 流内部有序;单个工具的分片有序)。 + +## 一轮内的可见顺序 + +一个带两次工具调用的轮,消费方按序观察到(标注示例): + +```text + 1 event request_begin + 2 partial partial_thinking(start → delta… → stop) + 3 complete thinking ← stop 后立即跟完整消息 + 4 partial partial_text(start → delta… → stop) + 5 complete text + 6 partial partial_tool_call A(start → delta… → stop) + 7 complete tool_call A + 8 event approval_decision(allow, A) ← 审批逐个,决策即产出;A 开始并发执行 + 9 partial partial_tool_call B(…) ← LLM 流继续,不等 A +10 complete tool_call B +11 event approval_decision(allow, B) +12 partial partial_tool_call_output B(…) ← B 先有输出:完成顺序,非调用顺序 +13 complete tool_call_output B +14 event token_usage ← LLM 流的最后一条 +15 event request_end(completed) ← LLM 流结束即产出,不等工具 +16 partial partial_tool_call_output A(…) ← 迟到输出出现在 request_end 之后 +17 complete tool_call_output A + (A、B 输出齐 → 按 A、B 原始顺序进入下一轮输入 → 下一个 request_begin) +``` + +若某条 `tool_call` 被拒绝,第 8 行的决策为 `deny`,随即产出一条合成的 `aborted` `tool_call_output`(内容 `Tool call denied by user.`),不派发执行。 + +## 顺序保证与非保证 + +**有保证:** + +| 保证 | 说明 | +| --- | --- | +| 分片纪律 | 每段严格 `start → delta* → stop`,完整消息紧随其后;全部 delta 拼接 ≡ 完整消息 | +| 审批位次 | `approval_decision` 在其 `tool_call` 之后、该工具任何输出之前 | +| 配对完整 | 每个已提交的 `tool_call` 恰好对应一条完整 `tool_call_output`(拒绝为合成输出) | +| LLM 流收尾 | `token_usage` 是 LLM 流的最后一条,`request_end` 紧随其后 | +| 提交判据 | `request_end.status === "completed"` ⇔ 该轮已被网关提交(回放据此取舍) | +| 流序 = Trace 序 | 逐条"边流边写";Trace 只是滤掉分片与 `origin` 消息 | +| 传输有序 | SSE 按通道单调 id 投递,断线按 `Last-Event-ID` 补发或 `resync_required`,见 [Server API](/server-api) | + +**无保证(渲染层不得依赖):** + +| 非保证 | 说明 | +| --- | --- | +| 工具输出顺序 | 到达顺序是完成顺序;多工具的分片可交错,须按 `tool_call_id` 归属 | +| `request_end` ≠ 轮结束 | 仍在执行的工具输出可出现在 `request_end` 之后、下一个 `request_begin` 之前 | +| 事件与内容的相对间隔 | `approval_decision` 与首条工具输出之间可能插入 LLM 流的后续消息 | + +## 流序与上下文序 + +同一批工具输出存在两种顺序,服务两个不同的消费者: + +- **流序(完成顺序)**——面向 Human:谁先完成谁先可见,保证实时性; +- **上下文序(原始调用顺序)**——面向模型:进入下一轮输入前按 `tool_call` 的原始顺序重排,保证与 Provider 的配对约定一致。 + +因此**渲染层不得用到达顺序重建上下文**——按 `tool_call_id` 把输出挂回对应调用即可;上下文顺序由 engine 负责。 + +## 边界情形的时序 + +| 情形 | 流上可见的顺序 | +| --- | --- | +| 用户中断 | (已产出的消息)→ `abort` 事件——`run` 返回前的最后一条;补发内容只进模型上下文,不上流、不进 Trace | +| 自动重连 | `request_end(timeout \| malformed)` → 新的 `request_begin`;`` 块仅模型可见 | +| 上下文压缩 | `compaction_begin` → 压缩请求在旧上下文中执行(其流式输出**不上行**,只写 Trace)→ 该请求的 `token_usage` → `compaction_end(status)` | +| 达到 max_turns | 长度提示消息 → 结束;未提交的输入按补发保留 | +| Prompt 本身 | 写入 Trace,但不回流(输入方已有) | +| session_meta | 主 Session 的输出流不产出它(存在于 Trace 与历史接口中);Subagent 子流的**第一条**是子 Session 的 `session_meta` | + +## 跨 Session:origin 链 + +`run_subagent` 派生的子 Session 有自己的完整消息流。转发给父级时,每条子消息的 `origin` 前插一跳子 Session id,与父级本地消息**按到达时刻交错**;渲染层按 `origin` 归入对应子会话卡片。子消息不写父 Trace——父 Trace 只保留 `subagent` 指针事件,子 Session 的流序记录在它自己的 Trace 里。 + +## 传输层顺序(SSE) + +Server 把上述输出流原样(单行 JSON)推入 per-Session SSE 通道:事件 id 单调递增,有界缓冲支持断线补发,重放窗口失效时以 `resync_required` 通知客户端重拉历史。事件次序:重连时补发的缺口(或 `resync_required`)在前,随后才是权威的 `task_state` 快照与未决审批;全新连接不重放缓冲,首条即为 `task_state` 快照。细节见 [Server API](/server-api) 的流式接口一节;自带 Web App 的"连接先行 + 去重"消费模式亦在该页。 diff --git a/packages/docs/content/models.en.md b/packages/docs/content/models.en.md new file mode 100644 index 0000000..418cfce --- /dev/null +++ b/packages/docs/content/models.en.md @@ -0,0 +1,88 @@ +--- +title: Models & Providers +description: Model access through the single AgentHub gateway, (provider, model_id) identity, the per-Project model table, credentials and thinking levels. +--- + +## One gateway + +All model access goes through one gateway library: `@prismshadow/agenthub` (AutoLLMClient). Core defines only a thin `LLMInterface` (see [Interfaces](/interfaces)); per-provider protocol adaptation happens inside AgentHub, so 1000+ online and local models are reachable, including any OpenAI-compatible endpoint. The protocol translation lives in `packages/core/src/llm/generative-model.ts`. + +## Model identity + +A model's identity is always the `(provider, model_id)` pair: `provider` is a config group name, `model_id` the upstream request id sent to AgentHub unchanged. The two are independent fields — concatenating them into one string is forbidden anywhere in the pipeline. + +## The per-Project model table + +Each Project's available models are recorded in the hidden `.project_config.toml`, maintained via the CLI (`penguin config model add / default / list`, see [CLI Reference](/cli)) or the Web UI — never hand-edited. `ModelEntry` fields: + +| Field | Meaning | +| --- | --- | +| `provider` | Config group name; paired with `model_id` it forms the unique key | +| `model_id` | Upstream request id | +| `context_window` | Context window | +| `client_type` | Protocol hint (e.g. `openai`); inferred by AgentHub from the model id when omitted | +| `display_name` | Display name | +| `vision` | Whether image input is supported, default true | +| `pricing` | Three price buckets (unit `usd_per_mtok`, USD per million tokens): `cache_read` / `cache_write` / `output` | +| `api_key` / `base_url` | Inlined credentials, both optional; when blank, AgentHub falls back to environment variables | + +A fresh Project defaults to deepseek-v4-pro. A `vision_model` entry can additionally designate the proxy model that `describe_image` uses for text-only session models (see [Tools & Approval](/tools)); it is unset by default. + +File shape (illustrative): + +```toml +default_model = { provider = "deepseek", model_id = "deepseek-v4-pro" } +vision_model = { provider = "google", model_id = "gemini-3.1-pro-preview" } + +[[models]] +provider = "deepseek" +model_id = "deepseek-v4-pro" +context_window = 1000000 + +[[models]] +provider = "custom" +model_id = "my-model" +client_type = "openai" +base_url = "https://llm.example.com/v1" +api_key = "sk-..." +``` + +For a model tagged `vision = false` (e.g. the DeepSeek series), images from conversation input are saved to the Session scratchpad and handed over as a file path spliced into the text, and the image-reading tool switches to `describe_image`. + +## Built-in provider groups + +Built-in groups and their env-var fallbacks (catalog source: `packages/core/src/state/model-catalog.ts`); each group also has a `_BASE_URL` variant (e.g. `ANTHROPIC_BASE_URL`): + +| Provider | API key env var | Notes | +| --- | --- | --- | +| deepseek | `DEEPSEEK_API_KEY` | Group of the default model | +| openrouter | `OPENAI_API_KEY` | OpenAI-compatible gateway, preset base URL `https://openrouter.ai/api/v1` | +| siliconflow | `OPENAI_API_KEY` | OpenAI-compatible gateway, preset base URL `https://api.siliconflow.cn/v1` | +| google | `GEMINI_API_KEY` | | +| anthropic | `ANTHROPIC_API_KEY` | | +| openai | `OPENAI_API_KEY` | | +| zhipu | `ZAI_API_KEY` | | +| moonshot | `MOONSHOT_API_KEY` | | +| custom | `OPENAI_API_KEY` | Any OpenAI-protocol endpoint | + +The gateway groups (openrouter / siliconflow) go through AgentHub's OpenAI client, so with blank credentials they read `OPENAI_API_KEY` — not a gateway-specific variable. + +Some models in the preset catalog: deepseek-v4-pro / deepseek-v4-flash, gemini-3.1-pro-preview, claude-opus-4-8 / claude-sonnet-4-6, gpt-5.5, glm-5.2, kimi-k2.6 (not exhaustive). + +## Thinking levels + +Five levels: `none | low | medium | high | xhigh`, configured per Agent as `model.thinking_level` in `system_config.yaml`, default medium. See [Configuration](/configuration). + +## Models decoupled from Agents + +An Agent never binds a model: the model is chosen when a Session is created and stays locked for that Session; the same Agent can run different Sessions on different models. The three `pricing` buckets feed the usage/cost center's per-Token accounting. + +Credential handling: + +- an inline `api_key` is stored in the hidden Project config file with mode 0600; +- the Web UI masks it on display; +- blank credentials fall back to the provider's environment variables. + +## Connectivity test + +The Web Models page offers a per-model connectivity test (owner only). diff --git a/packages/docs/content/models.zh.md b/packages/docs/content/models.zh.md new file mode 100644 index 0000000..db2c46e --- /dev/null +++ b/packages/docs/content/models.zh.md @@ -0,0 +1,88 @@ +--- +title: 模型与 Provider +description: 经 AgentHub 单一网关接入模型,以 (provider, model_id) 成对标识,Project 级模型表、凭证与思考等级配置。 +--- + +## 单一网关 + +所有模型访问都经由一个网关库:`@prismshadow/agenthub`(AutoLLMClient)。core 只定义一层很薄的 `LLMInterface`(见 [接口契约](/interfaces)),各 Provider 的协议适配全部由 AgentHub 完成,因此可以接入 1000+ 在线或本地模型,包括任意 OpenAI 兼容端点。协议翻译实现在 `packages/core/src/llm/generative-model.ts`。 + +## 模型标识 + +模型身份永远是 `(provider, model_id)` 成对表示:`provider` 是配置分组名,`model_id` 是原样发给上游的请求 id。二者是两个独立字段,任何环节都不允许拼接成一个字符串。 + +## Project 模型表 + +每个 Project 的可用模型记录在隐藏文件 `.project_config.toml` 中,由 CLI(`penguin config model add / default / list`,见 [CLI 参考](/cli))或 Web 界面维护,不手工编辑。`ModelEntry` 字段: + +| 字段 | 说明 | +| --- | --- | +| `provider` | 配置分组名,与 `model_id` 成对构成唯一键 | +| `model_id` | 上游请求 id | +| `context_window` | 上下文窗口 | +| `client_type` | 协议提示(如 `openai`);缺省由 AgentHub 按 model id 推断 | +| `display_name` | 显示名 | +| `vision` | 是否支持图像输入,默认 true | +| `pricing` | 三档价格(单位 `usd_per_mtok`,USD 每百万 Token):`cache_read` / `cache_write` / `output` | +| `api_key` / `base_url` | 内联凭证,可留空;留空时 AgentHub 回退读环境变量 | + +新建 Project 的默认模型是 deepseek-v4-pro。另可配置一条 `vision_model`,作为 text-only 模型使用 `describe_image` 时的代读模型(见 [工具与审批](/tools));默认不配置。 + +文件形态(示意): + +```toml +default_model = { provider = "deepseek", model_id = "deepseek-v4-pro" } +vision_model = { provider = "google", model_id = "gemini-3.1-pro-preview" } + +[[models]] +provider = "deepseek" +model_id = "deepseek-v4-pro" +context_window = 1000000 + +[[models]] +provider = "custom" +model_id = "my-model" +client_type = "openai" +base_url = "https://llm.example.com/v1" +api_key = "sk-..." +``` + +对标注 `vision = false` 的模型(如 DeepSeek 系列):对话输入中的图片会保存到 Session scratchpad,以文件路径形式拼入文本;读图工具切换为 `describe_image`。 + +## 内置 Provider 分组 + +内置分组及其环境变量回退(目录源:`packages/core/src/state/model-catalog.ts`);每个分组同时存在 `_BASE_URL` 变体(如 `ANTHROPIC_BASE_URL`): + +| Provider | API Key 环境变量 | 说明 | +| --- | --- | --- | +| deepseek | `DEEPSEEK_API_KEY` | 默认模型所在分组 | +| openrouter | `OPENAI_API_KEY` | OpenAI 兼容网关,预置 base URL `https://openrouter.ai/api/v1` | +| siliconflow | `OPENAI_API_KEY` | OpenAI 兼容网关,预置 base URL `https://api.siliconflow.cn/v1` | +| google | `GEMINI_API_KEY` | | +| anthropic | `ANTHROPIC_API_KEY` | | +| openai | `OPENAI_API_KEY` | | +| zhipu | `ZAI_API_KEY` | | +| moonshot | `MOONSHOT_API_KEY` | | +| custom | `OPENAI_API_KEY` | 任意 OpenAI 协议端点 | + +网关分组(openrouter / siliconflow)经 AgentHub 的 OpenAI 客户端请求,因此凭证留空时读取的是 `OPENAI_API_KEY`,而非网关自己的变量名。 + +预置目录中的部分模型:deepseek-v4-pro / deepseek-v4-flash、gemini-3.1-pro-preview、claude-opus-4-8 / claude-sonnet-4-6、gpt-5.5、glm-5.2、kimi-k2.6 等(非完整清单)。 + +## 思考等级 + +思考等级共五档:`none | low | medium | high | xhigh`,按 Agent 在 `system_config.yaml` 的 `model.thinking_level` 配置,默认 medium。见 [配置参考](/configuration)。 + +## 模型与 Agent 解耦 + +Agent 从不绑定模型:模型在创建 Session 时选定,并在该 Session 内锁定不变;同一个 Agent 可以在不同 Session 用不同模型运行。`pricing` 三档价格供用量/成本中心按 Token 计费。 + +凭证处理: + +- 内联 `api_key` 存放在权限 0600 的隐藏 Project 配置文件中; +- Web 界面展示时打码; +- 凭证留空时回退到对应 Provider 的环境变量。 + +## 连通性测试 + +Web 的模型页为每个模型提供连通性测试(仅 owner 可用)。 diff --git a/packages/docs/content/omni-message.en.md b/packages/docs/content/omni-message.en.md new file mode 100644 index 0000000..b54265d --- /dev/null +++ b/packages/docs/content/omni-message.en.md @@ -0,0 +1,297 @@ +--- +title: The OmniMessage Protocol +description: One envelope, three message types, a five-value stop_reason — the unified protocol behind the SDK, the Trace and SSE, field by field. +--- + +OmniMessage is PenguinHarness's unified message protocol: the SDK yields it, the Trace stores it line by line, and the Server pushes it verbatim over SSE. What streams, what is stored and what the model sees are one structure — there is no second format between front end, back end and storage. + +This page goes top-down: the envelope and the three message types first, then every payload field by field, then the protocol-wide semantics (streaming discipline, stop_reason, origin, fidelity fields). Type source: `packages/core/src/omnimessage/types.ts`. + +## The envelope + +Every message shares one envelope; only the `payload` varies: + +```ts +interface OmniMessage

{ + timestamp: string; // ISO 8601 UTC + type: "session_meta" | "model_msg" | "event_msg"; + payload: P; + origin?: string[]; // child-Session chain (outer→inner); absent = main Session +} +``` + +What each message type carries: + +| type | Meaning | Volume | +| --- | --- | --- | +| `session_meta` | The full runtime configuration of one model context | exactly one per context | +| `model_msg` | Content inside the model context (text, thinking, tool calls and results) | the bulk | +| `event_msg` | Runtime events outside the context (approvals, usage, compaction, aborts) | alongside | + +## session_meta + +```ts +interface SessionMetaPayload { + session_id: string; + provider: string; // one half of the model-identity pair + model_id: string; // the upstream request id sent to AgentHub + model_context_window: number | string; + system_prompt: string; // fully assembled, placeholders substituted + tools: ToolDefinition[]; // the complete tool schema sent to the model + thinking_level: string; // "default" when unconfigured + agent_state: string; // absolute path of the Agent State + workspace: string; // absolute path of the Workspace +} + +interface ToolDefinition { + name: string; + description: string; + parameters?: Record; // JSON Schema +} +``` + +On resume, the engine takes this Trace line as the runtime config — the model, system prompt and Workspace are immutable for the Session's lifetime. See [Sessions & Traces](/sessions-and-traces). + +## model_msg: complete payloads + +Seven content payloads, discriminated by `payload.type`. Shared optional fields: `stop_reason` (marks an abnormal terminal state) and `signature` (a provider-fidelity field, see below): + +```ts +interface TextPayload { + type: "text"; + role: "user" | "assistant"; + text: string; + phase?: string | null; // segmentation marker (e.g. GPT-5 phases) + signature?: string; + stop_reason?: StopReason; +} + +interface ThinkingPayload { + type: "thinking"; + role: "assistant"; + thinking: string; + signature?: string; // required by some models to replay history + stop_reason?: StopReason; +} + +interface InlineThinkingPayload { + type: "inline_thinking"; + role: "assistant"; + data: string; // reasoning content in binary form + mime_type: string; + signature?: string; + stop_reason?: StopReason; +} + +interface ToolCallPayload { + type: "tool_call"; + role: "assistant"; + name: string; + arguments: string; // arguments as a JSON string + tool_call_id: string; + signature?: string; + stop_reason?: StopReason; +} + +interface ToolCallOutputPayload { + type: "tool_call_output"; + role: "user"; + output: string; + images?: string[]; // data:;base64,… URLs (e.g. read_image results) + tool_call_id: string; + stop_reason?: StopReason; +} + +interface ImageUrlPayload { + type: "image_url"; + role: "user"; + image_url: string; // web URL or base64 data URL + stop_reason?: StopReason; +} + +interface InlineDataPayload { + type: "inline_data"; + role: "user" | "assistant"; + data: string; // other binary content + mime_type: string; + signature?: string; + stop_reason?: StopReason; +} +``` + +`tool_call` and `tool_call_output` pair strictly via `tool_call_id`; a turn's calls form one batch, and outputs are re-fed in the original call order (see [The Agent Loop](/agent-loop)). + +## model_msg: streaming partials + +Four `partial_*` payloads mirror their complete counterparts, carrying an `event_type` phase marker: + +```ts +type StreamEventType = "start" | "delta" | "stop"; + +interface PartialTextPayload { + type: "partial_text"; + role: "assistant"; + event_type: StreamEventType; + text: string; // the text added by this fragment + stop_reason?: StopReason; +} + +interface PartialThinkingPayload { + type: "partial_thinking"; + role: "assistant"; + event_type: StreamEventType; + thinking: string; + stop_reason?: StopReason; +} + +interface PartialToolCallPayload { + type: "partial_tool_call"; + role: "assistant"; + event_type: StreamEventType; + name: string; + arguments: string; // incremental fragment of the arguments JSON + tool_call_id: string; + stop_reason?: StopReason; +} + +interface PartialToolCallOutputPayload { + type: "partial_tool_call_output"; + role: "user"; + event_type: StreamEventType; + output: string; + images?: string[]; // images are not incremental — one delta carries the whole set + tool_call_id: string; + stop_reason?: StopReason; +} +``` + +### The streaming discipline + +Every streamed segment follows one timing rule, with the complete message immediately after the `stop`: + +```text +partial_text(start) → partial_text(delta) → … → partial_text(stop) → text (complete) + └── concatenation of all deltas ≡ the complete message ──┘ + (truncation applies to both alike) +``` + +Renderers can therefore paint deltas incrementally and swap in the complete message in place; the Trace records only complete messages, never fragments. Interface implementations close their structures internally and never leak an unclosed fragment upward. `PartialAggregator` (`aggregate.ts`) ships a ready-made aggregator. + +## event_msg + +Eight event payloads, all listed field by field: + +```ts +interface RequestBeginPayload { + type: "request_begin"; +} + +interface RequestEndPayload { + type: "request_end"; + status: StopReason; // "completed" is the mechanical commit criterion for replay +} + +interface ApprovalDecisionPayload { + type: "approval_decision"; + decision: "allow" | "deny"; + tool_call_id: string; // pairs with the approved tool_call — the audit record +} + +interface TokenUsagePayload { + type: "token_usage"; + session: TokenCounts; // Session cumulative + request: TokenCounts; // this Request +} + +interface TokenCounts { + cache_read: number; + cache_write: number; + output: number; + total: number; +} + +type CompactionReason = "context" | "turns" | "manual"; +type CompactionMode = "summarize" | "discard"; + +interface CompactionBeginPayload { + type: "compaction_begin"; + reason: CompactionReason; + mode: CompactionMode; + context: number; // context tokens at trigger time + turns: number; // cumulative turns at trigger time +} + +interface CompactionEndPayload { + type: "compaction_end"; + reason: CompactionReason; + mode: CompactionMode; + status: StopReason; +} + +interface AbortPayload { + type: "abort"; + reason?: string | null; +} + +interface SubagentPayload { + type: "subagent"; + session_id: string; // pointer in the parent Trace to a direct child Session +} +``` + +## stop_reason + +A five-value enum used across messages and interface results (`LLMOutcome.status` uses the same set — see [Core Interfaces](/interfaces)): + +```ts +type StopReason = "completed" | "failed" | "aborted" | "timeout" | "malformed"; +``` + +| Value | Meaning | Engine reaction | +| --- | --- | --- | +| `completed` | finished normally | continue | +| `aborted` | user interrupt | stop, hand back to the user | +| `timeout` | LLM timeout / lost connection | LLM side only: auto-reconnect within the run | +| `malformed` | parse failure / truncated stream | LLM side only: auto-reconnect within the run | +| `failed` | other non-retryable error | stop, hand back to the user | + +Errors never cross an interface boundary as exceptions — they *are* messages. See [The Agent Loop](/agent-loop). + +## origin: the Subagent chain + +`origin` serves Subagents: when a child Session's messages are forwarded to the parent, each hop prepends one child Session id (outer→inner), and renderers route messages into the right nested card by the chain: + +```ts +// message from the main Session: no origin +{ timestamp: "…", type: "model_msg", payload: { type: "text", … } } + +// message from a one-level Subagent: origin = [child Session id] +{ timestamp: "…", type: "model_msg", origin: ["session-2026-07-18-…-a1b2c3d4"], payload: { … } } +``` + +`origin`-tagged messages are not written to the parent Trace — the child Session has its own Trace, and the parent keeps only the `subagent` pointer event. + +## Provider-fidelity fields + +Provider-specific fields such as `signature` and `phase` pass through and persist verbatim end to end — some models require them byte-for-byte when history is replayed, and any rewriting would break compatibility. This is one of the preconditions for lossless Session recovery from the Trace. + +## Three jobs, one protocol + +| Surface | Subset used | +| --- | --- | +| SDK boundary (`session.run` output) | complete `model_msg` + streaming `partial_*` + all `event_msg` | +| Trace on disk | `session_meta` + complete `model_msg` + all `event_msg` (no partials, no `origin`-tagged messages) | +| Server SSE stream | same as the SDK boundary, verbatim single-line JSON — see [Server API](/server-api) | + +How messages travel along these surfaces — and every ordering guarantee — is covered on [Message Flow & Ordering](/message-flow). + +## Builders and guards + +`@prismshadow/penguin-core` exports all types, a builder per message kind (`builders.ts`: `userText`, `assistantText`, `toolCall`, `toolCallOutput`, `partialText`, `tokenUsage`, `withOrigin`, `emptyTokenCounts`, `addTokenCounts`, …) and runtime guards (`isCompleteModelMessage`, `isPartialPayload`, `isModelMessage`, `isEventMessage`, `isSessionMeta`): + +```ts +import { userText, isCompleteModelMessage } from "@prismshadow/penguin-core"; + +const prompt = userText("List the files in the current directory"); +// { timestamp: "…", type: "model_msg", payload: { type: "text", role: "user", text: "…" } } +``` diff --git a/packages/docs/content/omni-message.zh.md b/packages/docs/content/omni-message.zh.md new file mode 100644 index 0000000..5f659c2 --- /dev/null +++ b/packages/docs/content/omni-message.zh.md @@ -0,0 +1,296 @@ +--- +title: OmniMessage 协议 +description: 一个信封、三类消息、五值 stop_reason——贯穿 SDK、Trace 与 SSE 的统一消息协议,逐字段定义。 +--- + +OmniMessage 是 PenguinHarness 的统一消息协议:SDK 对外产出它,Trace 逐行存储它,Server 经 SSE 原样推送它。「流出去的」「存下来的」「模型看到的」是同一种结构,前后端与存储之间不存在第二套格式。 + +本页自顶向下:先定义信封与三类消息,再逐字段展开每一种 payload,最后是贯穿全协议的语义(流式纪律、stop_reason、origin、保真字段)。类型源码:`packages/core/src/omnimessage/types.ts`。 + +## 信封 + +所有消息共享同一个信封,仅 `payload` 不同: + +```ts +interface OmniMessage

{ + timestamp: string; // ISO 8601 UTC + type: "session_meta" | "model_msg" | "event_msg"; + payload: P; + origin?: string[]; // 子 Session 链(由外到内);缺省表示主 Session +} +``` + +三类消息的分工: + +| type | 含义 | 数量级 | +| --- | --- | --- | +| `session_meta` | 一个模型上下文的完整运行配置 | 每个上下文恰好一条 | +| `model_msg` | 模型上下文中的内容消息(文本、思考、工具调用与结果) | 主体 | +| `event_msg` | 上下文之外的运行事件(审批、用量、压缩、中断) | 伴随 | + +## session_meta + +```ts +interface SessionMetaPayload { + session_id: string; + provider: string; // 模型身份二元组之一 + model_id: string; // 发给 AgentHub 的上游请求 id + model_context_window: number | string; + system_prompt: string; // 占位符替换完成后的完整系统提示词 + tools: ToolDefinition[]; // 发给模型的完整工具 schema + thinking_level: string; // 未配置时为 "default" + agent_state: string; // Agent State 绝对路径 + workspace: string; // Workspace 绝对路径 +} + +interface ToolDefinition { + name: string; + description: string; + parameters?: Record; // JSON Schema +} +``` + +恢复 Session 时,引擎直接以 Trace 中的这条消息为运行时配置——模型、系统提示词、Workspace 在 Session 生命周期内不可变,见 [Session 与 Trace](/sessions-and-traces)。 + +## model_msg:完整消息 + +七种内容 payload,以 `payload.type` 判别。公共可选字段:`stop_reason`(非正常收尾时标注终态)与 `signature`(Provider 保真字段,见下文): + +```ts +interface TextPayload { + type: "text"; + role: "user" | "assistant"; + text: string; + phase?: string | null; // 分段标记(如 GPT-5 的 phase) + signature?: string; + stop_reason?: StopReason; +} + +interface ThinkingPayload { + type: "thinking"; + role: "assistant"; + thinking: string; + signature?: string; // 部分模型历史回放所必需 + stop_reason?: StopReason; +} + +interface InlineThinkingPayload { + type: "inline_thinking"; + role: "assistant"; + data: string; // 二进制形态的思考内容 + mime_type: string; + signature?: string; + stop_reason?: StopReason; +} + +interface ToolCallPayload { + type: "tool_call"; + role: "assistant"; + name: string; + arguments: string; // 参数 JSON 字符串 + tool_call_id: string; + signature?: string; + stop_reason?: StopReason; +} + +interface ToolCallOutputPayload { + type: "tool_call_output"; + role: "user"; + output: string; + images?: string[]; // data:;base64,… 列表(如 read_image 的结果) + tool_call_id: string; + stop_reason?: StopReason; +} + +interface ImageUrlPayload { + type: "image_url"; + role: "user"; + image_url: string; // 网络 URL 或 base64 data URL + stop_reason?: StopReason; +} + +interface InlineDataPayload { + type: "inline_data"; + role: "user" | "assistant"; + data: string; // 其他二进制内容 + mime_type: string; + signature?: string; + stop_reason?: StopReason; +} +``` + +`tool_call` 与 `tool_call_output` 通过 `tool_call_id` 严格配对;一轮内的多个调用是一个批次,输出按原始调用顺序回填(见 [Agent 运行循环](/agent-loop))。 + +## model_msg:流式分片 + +四种 `partial_*` payload 与完整消息一一对应,携带 `event_type` 标记分片阶段: + +```ts +type StreamEventType = "start" | "delta" | "stop"; + +interface PartialTextPayload { + type: "partial_text"; + role: "assistant"; + event_type: StreamEventType; + text: string; // 本条分片新增的文本 + stop_reason?: StopReason; +} + +interface PartialThinkingPayload { + type: "partial_thinking"; + role: "assistant"; + event_type: StreamEventType; + thinking: string; + stop_reason?: StopReason; +} + +interface PartialToolCallPayload { + type: "partial_tool_call"; + role: "assistant"; + event_type: StreamEventType; + name: string; + arguments: string; // 参数 JSON 的增量片段 + tool_call_id: string; + stop_reason?: StopReason; +} + +interface PartialToolCallOutputPayload { + type: "partial_tool_call_output"; + role: "user"; + event_type: StreamEventType; + output: string; + images?: string[]; // 图像不增量,由单条 delta 整体携带 + tool_call_id: string; + stop_reason?: StopReason; +} +``` + +### 流式纪律 + +每段流式内容严格遵守同一时序,`stop` 之后立即跟随对应的完整消息: + +```text +partial_text(start) → partial_text(delta) → … → partial_text(stop) → text(完整) + └── 全部 delta 拼接 ≡ 完整消息内容(截断也两侧同步) ──┘ +``` + +因此渲染层可以先增量渲染、收到完整消息后原地替换;Trace 只记录完整消息,不存分片。接口实现方在内部把结构闭合完毕,永远不向上层泄漏未闭合的分片。`PartialAggregator`(`aggregate.ts`)提供现成的分片聚合实现。 + +## event_msg + +八种事件 payload,全部逐字段列出: + +```ts +interface RequestBeginPayload { + type: "request_begin"; +} + +interface RequestEndPayload { + type: "request_end"; + status: StopReason; // completed 是回放判定「该轮已提交」的机械标准 +} + +interface ApprovalDecisionPayload { + type: "approval_decision"; + decision: "allow" | "deny"; + tool_call_id: string; // 与被审批的 tool_call 配对,构成审计记录 +} + +interface TokenUsagePayload { + type: "token_usage"; + session: TokenCounts; // Session 累计 + request: TokenCounts; // 本次 Request +} + +interface TokenCounts { + cache_read: number; + cache_write: number; + output: number; + total: number; +} + +type CompactionReason = "context" | "turns" | "manual"; +type CompactionMode = "summarize" | "discard"; + +interface CompactionBeginPayload { + type: "compaction_begin"; + reason: CompactionReason; + mode: CompactionMode; + context: number; // 触发时的上下文 Token 数 + turns: number; // 触发时的累计轮数 +} + +interface CompactionEndPayload { + type: "compaction_end"; + reason: CompactionReason; + mode: CompactionMode; + status: StopReason; +} + +interface AbortPayload { + type: "abort"; + reason?: string | null; +} + +interface SubagentPayload { + type: "subagent"; + session_id: string; // 父 Trace 中指向直接子 Session 的指针 +} +``` + +## stop_reason + +五值枚举,贯穿消息与接口返回(`LLMOutcome.status` 使用同一集合,见[接口契约](/interfaces)): + +```ts +type StopReason = "completed" | "failed" | "aborted" | "timeout" | "malformed"; +``` + +| 值 | 语义 | 引擎的反应 | +| --- | --- | --- | +| `completed` | 正常完成 | 继续 | +| `aborted` | 用户中断 | 停止并交还用户 | +| `timeout` | LLM 超时/断连 | 仅 LLM 侧:同一 run 内自动重连 | +| `malformed` | 响应解析失败/流截断 | 仅 LLM 侧:同一 run 内自动重连 | +| `failed` | 其他不可重试错误 | 停止并交还用户 | + +错误从不以异常形式穿过接口边界——它们就是消息,见 [Agent 运行循环](/agent-loop)。 + +## origin:子 Session 链 + +`origin` 服务于 Subagent:子 Session 的消息转发给父级时,每经过一层就在数组前端添加一个子 Session id(由外到内),渲染层据此把消息归入对应的子会话卡片: + +```ts +// 主 Session 的消息:无 origin +{ timestamp: "…", type: "model_msg", payload: { type: "text", … } } + +// 一层 Subagent 的消息:origin = [子 Session id] +{ timestamp: "…", type: "model_msg", origin: ["session-2026-07-18-…-a1b2c3d4"], payload: { … } } +``` + +带 `origin` 的消息不写入父 Trace——子 Session 拥有自己的 Trace,父 Trace 只保留 `subagent` 指针事件。 + +## 保真字段 + +`signature` 与 `phase` 等 Provider 专有字段在整条链路上原样透传、原样存储——部分模型在历史回放时要求逐字一致,任何转写都会破坏兼容性。这是 Trace 能够无损恢复 Session 的前提之一。 + +## 协议的三种职责 + +| 场景 | 使用的子集 | +| --- | --- | +| SDK 边界(`session.run` 输出) | 完整 `model_msg` + 流式 `partial_*` + 全部 `event_msg` | +| Trace 落盘 | `session_meta` + 完整 `model_msg` + 全部 `event_msg`(不存分片与 `origin` 消息) | +| Server SSE 推送 | 与 SDK 边界一致,原样单行 JSON,见 [Server API](/server-api) | + +消息沿这些通道传递的机制与顺序保证,见[消息流转与时序](/message-flow)。 + +## 构造与判别 + +`@prismshadow/penguin-core` 导出全部类型、每种消息的构造函数(`builders.ts`:`userText`、`assistantText`、`toolCall`、`toolCallOutput`、`partialText`、`tokenUsage`、`withOrigin`、`emptyTokenCounts`、`addTokenCounts` 等)与运行时判别函数(`isCompleteModelMessage`、`isPartialPayload`、`isModelMessage`、`isEventMessage`、`isSessionMeta`): + +```ts +import { userText, isCompleteModelMessage } from "@prismshadow/penguin-core"; + +const prompt = userText("列出当前目录的文件"); +// { timestamp: "…", type: "model_msg", payload: { type: "text", role: "user", text: "…" } } +``` diff --git a/packages/docs/content/quickstart.en.md b/packages/docs/content/quickstart.en.md new file mode 100644 index 0000000..f5b6342 --- /dev/null +++ b/packages/docs/content/quickstart.en.md @@ -0,0 +1,76 @@ +--- +title: Quickstart +description: Install PenguinHarness, configure a model, and run your first Task. +--- + +## Install + +One-liner for Linux / macOS: + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +``` + +For other options (npm, from source), see [Installation](/installation). + +## Configure a model + +PenguinHarness ships with no built-in model credentials, so configure a model first. Use the Models page in the Web UI, or the CLI: + +```bash +penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default +``` + +- When `--provider` is omitted, the Provider is inferred from the built-in catalog. +- The API key can also come from environment variables: when a model entry has no inline api_key, AgentHub (the LLM gateway library) reads variables such as `DEEPSEEK_API_KEY`, `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, and `GEMINI_API_KEY`. A `.env` file in the working directory is loaded automatically. + +## Start the Web App + +```bash +penguin web +``` + +The service runs at http://127.0.0.1:7364 and opens your browser (`--no-open` to skip). First login is `admin` / `admin123` — change it right away. `penguin server` starts the same process headless. + +## One-shot run + +```bash +penguin run -m "Create hello.txt containing Hello, Penguin" +``` + +The Workspace defaults to the current directory; pass `--workspace /path` to change it. The target directory must already exist. + +## Interactive chat + +```bash +penguin chat +``` + +- Each input line starts a Task. +- `/compact` compacts the context; `/exit` or `/quit` quits; Ctrl-C interrupts the running Task. +- On exit it prints a `penguin chat --resume ` hint for resuming this Session; `--resume` without an id resumes the Agent's latest Session. + +## SDK hello + +After installing `@prismshadow/penguin-core`: + +```ts +import { createAgent, isCompleteModelMessage, userText } from "@prismshadow/penguin-core"; + +const agent = await createAgent({ agentId: "default_agent" }); +const session = await agent.createSession({ workspaceDir: process.cwd() }); + +for await (const output of session.run([userText("Create hello.txt containing hi")], { + approve: async () => "allow", +})) { + if (isCompleteModelMessage(output) && output.payload.type === "text") { + console.log(output.payload.text); + } +} +``` + +## Next steps + +- [Web App Guide](/web-app): use PenguinHarness from the browser. +- [CLI Reference](/cli): the full list of commands and options. +- [Architecture Overview](/architecture): how the pieces fit together. diff --git a/packages/docs/content/quickstart.zh.md b/packages/docs/content/quickstart.zh.md new file mode 100644 index 0000000..c5e28c8 --- /dev/null +++ b/packages/docs/content/quickstart.zh.md @@ -0,0 +1,76 @@ +--- +title: 快速开始 +description: 安装 PenguinHarness、配置模型并运行第一个 Task。 +--- + +## 安装 + +Linux / macOS 一键安装: + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +``` + +其他方式(npm、源码)见[安装](/installation)。 + +## 配置模型 + +PenguinHarness 不内置任何模型凭据,使用前需要先配置一个模型。可以在 Web UI 的 Models 页面完成,也可以用 CLI: + +```bash +penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default +``` + +- 省略 `--provider` 时,根据内置目录自动推断 Provider。 +- API Key 也可以来自环境变量:当模型条目没有内联 api_key 时,LLM 网关库 AgentHub 会读取 `DEEPSEEK_API_KEY`、`ANTHROPIC_API_KEY`、`OPENAI_API_KEY`、`GEMINI_API_KEY` 等变量;工作目录下的 `.env` 会被自动加载。 + +## 启动 Web App + +```bash +penguin web +``` + +服务运行在 http://127.0.0.1:7364 并自动打开浏览器(`--no-open` 跳过)。首次登录使用 `admin` / `admin123`,请立即修改密码。`penguin server` 启动同一进程的 headless 版本。 + +## 单次运行 + +```bash +penguin run -m "创建 hello.txt,内容为 Hello, Penguin" +``` + +Workspace 默认为当前目录,可用 `--workspace /path` 指定;目标目录必须已存在。 + +## 交互式对话 + +```bash +penguin chat +``` + +- 每输入一行即发起一个 Task。 +- `/compact` 压缩上下文;`/exit` 或 `/quit` 退出;Ctrl-C 中断正在运行的 Task。 +- 退出时会打印 `penguin chat --resume ` 提示,用于恢复本次 Session;`--resume` 不带 id 时恢复该 Agent 最近的 Session。 + +## SDK 示例 + +安装 `@prismshadow/penguin-core` 后: + +```ts +import { createAgent, isCompleteModelMessage, userText } from "@prismshadow/penguin-core"; + +const agent = await createAgent({ agentId: "default_agent" }); +const session = await agent.createSession({ workspaceDir: process.cwd() }); + +for await (const output of session.run([userText("Create hello.txt containing hi")], { + approve: async () => "allow", +})) { + if (isCompleteModelMessage(output) && output.payload.type === "text") { + console.log(output.payload.text); + } +} +``` + +## 下一步 + +- [Web App 指南](/web-app):在浏览器中使用 PenguinHarness。 +- [CLI 参考](/cli):完整命令与选项。 +- [架构总览](/architecture):了解整体设计。 diff --git a/packages/docs/content/self-improvement.en.md b/packages/docs/content/self-improvement.en.md new file mode 100644 index 0000000..2c7095a --- /dev/null +++ b/packages/docs/content/self-improvement.en.md @@ -0,0 +1,73 @@ +--- +title: Self-Improvement +description: The Skill-orchestrated Benchmark and optimization loop: score, improve, snapshot, roll back. +--- + +Self-improvement in PenguinHarness is not carried by special-purpose engine code — it is carried by Skills orchestrating the ordinary Agent machinery: evaluations are ordinary Sessions, optimization is ordinary file editing, and orchestration uses the built-in `run_subagent` tool. The direct payoff is that the whole process shares the same observability and recovery machinery as everyday runs. + +## Three roles + +| Role | Responsibility | +| --- | --- | +| Target Agent | The Agent being improved; runs evaluation tasks only inside its own Workspace | +| Evaluator | Runs and scores one Benchmark Case run | +| Optimizer | Drives the whole optimization loop | + +The roles are defined by Skills, not hardcoded: the Evaluator follows the `agent-evaluation` Skill, the Optimizer follows the `agent-optimization` Skill. This applies the design principle stated in the [Configuration Reference](/configuration) — an Agent's behavior is editable files on disk, which is what makes Agents improvable by Agents. + +## The loop + +1. `benchmark-design` builds a multi-Case capability Benchmark: repeated independent runs, with a traceable baseline calibrated first; +2. The Optimizer orchestrates Evaluators in parallel via the `run_subagent` tool, covering the Case × runs matrix; +3. Scores plus their linked Traces show where points were lost; +4. The Optimizer edits the Target Agent's editable state — `AGENTS.md`, Skills, config — to produce version N+1; +5. A Snapshot is taken before each round; the candidate version is kept only if the total score strictly improves, otherwise rolled back. + +Benchmark optimization mode requires a complete baseline series in the scoreboard — without a calibrated baseline there is no improvement to compare against. Besides this loop, `agent-optimization` also supports a one-shot feedback mode: a concrete correction is applied directly as edits to the Target Agent's state, without going through the evaluation loop. + +## Benchmark storage + +Benchmarks are stored per Agent under `benchmarks//`: + +```text +benchmarks// +├── benchmark_config.toml # Benchmark configuration (e.g. runs per Case) +├── / +│ ├── statement/ # the task given to the Target Agent +│ └── rubric/ # private scoring rubric, isolated from the Target Agent +└── scoreboard.yaml # evaluation records (v2 format) +``` + +The separation of `rubric/` from `statement/` is deliberate: the Target Agent sees only the task statement and never touches the scoring rubric. + +Each evaluation record in `scoreboard.yaml` (v2 format) is timestamped and carries: + +- the paired model reference `(provider, model_id)` used for the round; +- `summary_title` and `summary` (the round's conclusion and the hypothesis for the next one); +- total score, cost, and duration — Case-level metrics are the average over its runs, evaluation-level metrics are the sum over its Cases; +- per-Case run details, each run recording `score`, `cost`, `duration_ms`, and `session_id`. + +The built-in `default_agent` ships with an example Benchmark (`packages/core/src/state/example-benchmark.ts`) so the evaluation pages have data out of the box; the whole directory can be deleted or replaced at any time. + +## Snapshots and versions + +Before each optimization round, the Agent State is packed into `snapshots/v.tar.gz` (excluding the Vault — secrets never enter a snapshot). The `version` in `system_config.yaml` increments on successful optimization. The Web UI supports exporting and importing snapshots; importing a version not higher than the current one requires explicit confirmation. + +## Auditable end to end + +- Every Evaluator run is an ordinary Session with a full Trace; +- Scoreboard records link back to those Sessions via `session_id`; see [Sessions & Traces](/sessions-and-traces); +- The Web evaluation pages are read-only views of these files; see the [Web App Guide](/web-app). + +Scores are not black-box output: every number can be traced back to the run that produced it. + +## Related Skills + +| Skill | Purpose | +| --- | --- | +| `agent-creation` | Turn a requirement into a working Agent: write its `AGENTS.md`, install the Skills it needs | +| `benchmark-design` | Design and calibrate a multi-Case capability Benchmark | +| `agent-evaluation` | Run and score one isolated Benchmark Case run | +| `agent-optimization` | Improve an Agent from feedback or Benchmark results | + +How Skills are organized and installed is covered in the [Skill System](/skills). diff --git a/packages/docs/content/self-improvement.zh.md b/packages/docs/content/self-improvement.zh.md new file mode 100644 index 0000000..02bccfe --- /dev/null +++ b/packages/docs/content/self-improvement.zh.md @@ -0,0 +1,73 @@ +--- +title: 自我进化 +description: 由 Skill 编排的 Benchmark 评测与优化闭环:评分、改进、Snapshot 与回滚。 +--- + +PenguinHarness 中的自我进化不依赖专用引擎代码,而是由 Skill 编排普通的 Agent 机制完成:评测是普通的 Session,优化是普通的文件编辑,编排靠内置的 `run_subagent` 工具。这样做的直接收益是——整个过程与日常运行共用同一套可观测性与恢复机制。 + +## 三个角色 + +| 角色 | 职责 | +| --- | --- | +| Target Agent | 被改进的 Agent,只在自己的 Workspace 里执行评测任务 | +| Evaluator | 执行并评分一次 Benchmark Case 运行 | +| Optimizer | 驱动整个优化循环 | + +角色由 Skill 定义而非硬编码:Evaluator 遵循 `agent-evaluation` Skill,Optimizer 遵循 `agent-optimization` Skill。这正是[配置参考](/configuration)所述设计原则的应用——Agent 的行为是磁盘上的可编辑文件,所以 Agent 可以被 Agent 改进。 + +## 优化循环 + +1. `benchmark-design` 构建多 Case 的能力 Benchmark:重复独立运行,先校准出可追溯的基线; +2. Optimizer 通过 `run_subagent` 工具并行编排 Evaluator,覆盖 Case × 运行次数矩阵; +3. 得分与其关联的 Trace 共同指出失分位置; +4. Optimizer 编辑 Target Agent 的可编辑状态——`AGENTS.md`、Skills、配置——产出版本 N+1; +5. 每轮开始前先打 Snapshot;总分严格提升才保留候选版本,否则回滚。 + +Benchmark 优化模式要求 scoreboard 中已有完整的基线序列——没有校准过的基线,就没有可比较的提升。除此之外 `agent-optimization` 还支持一次性反馈模式:把一条具体的纠正意见直接落实为对 Target Agent 状态的编辑,不经过评测循环。 + +## Benchmark 存储 + +Benchmark 按 Agent 存放在 `benchmarks//` 下: + +```text +benchmarks// +├── benchmark_config.toml # Benchmark 配置(如每个 Case 的运行次数 runs) +├── / +│ ├── statement/ # 交给 Target Agent 的任务描述 +│ └── rubric/ # 私有评分标准,对 Target Agent 隔离 +└── scoreboard.yaml # 评测记录(v2 格式) +``` + +`rubric/` 与 `statement/` 的隔离是刻意设计:Target Agent 只能看到题面,永远接触不到评分标准。 + +`scoreboard.yaml`(v2 格式)中的每条评测记录带时间戳,并记录: + +- 本轮使用的模型成对引用 `(provider, model_id)`; +- `summary_title` 与 `summary`(本轮结论与下一轮假设); +- 总分、成本与耗时——Case 级指标是各次运行的平均值,评测级指标是各 Case 的加和; +- 每个 Case 的逐次运行明细,每次运行含 `score`、`cost`、`duration_ms` 与 `session_id`。 + +内置的 `default_agent` 预置了一个示例 Benchmark(`packages/core/src/state/example-benchmark.ts`),评测页面开箱即有数据;整个目录可随时删除或替换。 + +## Snapshot 与版本 + +每轮优化前,Agent State 被打包为 `snapshots/v.tar.gz`(Vault 除外——密钥永不进入快照)。`system_config.yaml` 的 `version` 在优化成功后自增。Web UI 支持导出与导入快照,导入版本不高于当前版本时需要显式确认。 + +## 全程可审计 + +- 每次 Evaluator 运行都是一个普通的 Session,留有完整 Trace; +- scoreboard 记录通过 `session_id` 链接回这些 Session,见 [Session 与 Trace](/sessions-and-traces); +- Web 的评测页面是这些文件的只读视图,见 [Web App 指南](/web-app)。 + +分数不是黑盒输出:任何一个数字都可以回溯到产生它的那次运行。 + +## 相关 Skill + +| Skill | 用途 | +| --- | --- | +| `agent-creation` | 把需求变成可用的 Agent:撰写其 `AGENTS.md`、安装所需 Skill | +| `benchmark-design` | 设计并校准多 Case 的能力 Benchmark | +| `agent-evaluation` | 隔离执行并评分一次 Benchmark Case 运行 | +| `agent-optimization` | 根据反馈或 Benchmark 结果改进 Agent | + +Skill 的组织与安装方式见[技能系统](/skills)。 diff --git a/packages/docs/content/server-api.en.md b/packages/docs/content/server-api.en.md new file mode 100644 index 0000000..7182536 --- /dev/null +++ b/packages/docs/content/server-api.en.md @@ -0,0 +1,236 @@ +--- +title: Server API +description: HTTP API reference — authentication, routes, the SSE streaming protocol, and DTO type imports. +--- + +The PenguinHarness server exposes a same-origin HTTP API used by the bundled Web App and by any other HTTP client. This page is the reference: authentication, route tables, and the SSE streaming protocol. For starting the server, see the [Quickstart](/quickstart). + +## Overview + +- Stack: Hono + @hono/node-server, requires Node >= 24; +- Storage: SQLite (built-in `node:sqlite`, WAL mode) holds only indexes and aggregates — users, auth sessions, Project authorization, Agent / Session indexes, usage, UI preferences, error records, and Schedule state; all Agent, Trace, and Workspace data stays as files under `~/.penguin/data`, shared with the CLI / SDK — see the [Configuration Reference](/configuration); +- Binding: defaults to `127.0.0.1:7364`, adjustable via the `PORT` / `HOST` environment variables; +- Request bodies: writes accept JSON only (Content-Type check, one of the CSRF defenses), capped at 20MB; +- Errors share a single shape: + +```text +{ "error": { "code": "", "message": "" } } +``` + +## Source layout + +```text +packages/server/src +├── index.ts / config.ts / app.ts # startup entry · env config · Hono assembly (createApp binds no port — testable) +├── api/types.ts # the outward DTO contract (type-only import via the "./api" subpath) +├── auth/ # scrypt passwords, admin seeding, cookie sessions, auth middleware +├── db/ # node:sqlite connection, schema SQL, one repo per table +├── http/ # error bodies, request validation, SSE adapter, routes/ all route groups +├── runtime/ # session-manager (runtime driving) · channel (SSE ring buffer) +│ # approvals · usage-recorder · scheduler · title-generator +└── services/ # authorization rules, TOML/YAML config IO, Session/Trace/usage/snapshot services +``` + +## Authentication + +- Cookie session: `penguin_session` (HttpOnly, SameSite=Lax), valid for 7 days with sliding renewal; +- Passwords are stored as scrypt hashes; the server keeps only the sha256 of the session token, never the plaintext; +- No open registration: the built-in admin `admin` / `admin123` is seeded at startup, and all other accounts are created by an admin; +- Same-origin only — no CORS middleware is enabled. + +```bash +curl -c cookies.txt -H "Content-Type: application/json" \ + -d '{"userId":"admin","password":"admin123"}' \ + http://127.0.0.1:7364/api/auth/login +``` + +## Route Reference + +### Auth and Account + +| Method | Path | Description | +| --- | --- | --- | +| POST | /api/auth/login | Log in: `{userId, password}` → `{user}` | +| POST | /api/auth/logout | Log out, returns 204 | +| GET | /api/me | Current user info | +| PUT | /api/me/password | Change password: `{oldPassword, newPassword}` | +| GET | /api/me/prefs | Read UI preferences | +| PUT | /api/me/prefs | Write UI preferences (shallow merge) | + +### User Administration (admin only) + +| Method | Path | Description | +| --- | --- | --- | +| GET | /api/admin/users | List users | +| POST | /api/admin/users | Create a user: `{userId, password}` | +| POST | /api/admin/users/:userId/password | Reset a password (invalidates all of that user's login sessions) | +| DELETE | /api/admin/users/:userId | Delete a user | + +### Projects and Members + +| Method | Path | Description | +| --- | --- | --- | +| GET | /api/projects | Projects visible to the current user | +| POST | /api/projects | Create a Project | +| DELETE | /api/projects/:projectId | Delete a Project | +| GET | /api/projects/:projectId/members | List members | +| POST | /api/projects/:projectId/members | Add a member: `{userId}` | +| DELETE | /api/projects/:projectId/members/:userId | Remove a member | + +Member writes are owner-only. + +### Models + +| Method | Path | Description | +| --- | --- | --- | +| GET | /api/projects/:projectId/models | List models (api_key masked) | +| PUT | /api/projects/:projectId/models | Full-table replace, keyed by `(provider, modelId)` | +| POST | /api/projects/:projectId/models/test | Connectivity test: `{provider, modelId, …}` → `{ok, latencyMs?, message?}` | + +### Agents + +The paths below omit the `/api/projects/:projectId` prefix. + +| Method | Path | Description | +| --- | --- | --- | +| GET / POST | /agents | List / create Agents | +| DELETE | /agents/:agentId | Delete an Agent | +| GET / PUT | /agents/:agentId/config | Read / write config (AGENTS.md + system_config.yaml; PUT preserves YAML comments) | +| GET / PUT | /agents/:agentId/vault | Vault environment variables (values masked; PUT is a full replace) | +| GET | /agents/:agentId/export | Export the Agent State snapshot (tar.gz download) | +| POST | /agents/:agentId/import | Import a snapshot: `{dataBase64, confirm?}`; 409 on version conflict without confirm | +| GET / POST | /agents/:agentId/skills | List / install installed Skills | +| DELETE | /agents/:agentId/skills/:name | Uninstall a Skill | +| GET | /agents/:agentId/benchmarks | Benchmark scoring data (read-only) | + +### Schedules + +| Method | Path | Description | +| --- | --- | --- | +| GET / POST | /agents/:agentId/schedules | List scheduled tasks / create one (409 if the name exists) | +| GET / PUT / DELETE | /agents/:agentId/schedules/:name | Read / update / delete a single task | + +Schedule writes are owner-only. + +### Session Creation and Directory Browsing + +| Method | Path | Description | +| --- | --- | --- | +| GET | /agents/:agentId/sessions | List Sessions (including run state) | +| POST | /agents/:agentId/sessions | Create a Session: `{modelId?, provider?, workspace?, approvalMode?}` → 201 | +| GET | /dirs?path= | Server-side directory browser (backs the Workspace picker) | + +On Session creation, the model defaults to the Project's default model, the Workspace defaults to an auto-created temporary directory, and the approval mode defaults to `allow-all`. + +### Usage and Traces (Agent Level) + +| Method | Path | Description | +| --- | --- | --- | +| GET | /usage | Usage statistics; query parameters `from`, `to`, `groupBy`, `agentId`, `provider`, `modelId` | +| GET | /agents/:agentId/traces | Date → Session drill-down structure of Trace files | +| GET | /agents/:agentId/traces/:sessionId/:index | Read Trace events (`offset` / `limit` pagination) | +| GET | /agents/:agentId/traces/:sessionId/:index/analysis | Trace performance analysis | + +### Session-Level Endpoints + +The paths below omit the `/api/sessions/:sessionId` prefix. For the storage model behind Sessions and Traces, see [Sessions and Traces](/sessions-and-traces). + +| Method | Path | Description | +| --- | --- | --- | +| GET | / | Session info | +| PATCH | / | Update: `{approvalMode?, archived?, title?}` | +| DELETE | / | Delete the Session (along with its Traces and scratch files) | +| GET | /messages | Full OmniMessage history | +| GET | /stream | SSE event stream (next section) | +| POST | /tasks | Start a Task: `{input: TaskInputPart[]}` → 202 | +| POST | /approvals/:toolCallId | Approval decision: `{decision}` is `allow` or `deny` → 204 | +| POST | /abort | Interrupt the current Task: 202 when triggered, 204 when idle | +| POST | /compact | Trigger context compaction: 202; 409 `nothing_to_compact` when there is nothing to compact | +| GET | /files?path= | Browse the Workspace directory | +| GET | /files/content?path=&download= | Read a Workspace file (`download=1` serves it as an attachment) | +| POST | /files/stat | Batch existence check: `{paths}` | +| PUT | /files/content?path= | Upload a file: `{dataBase64}`, capped at 14MB | +| GET | /traces | List this Session's Trace files | +| GET | /traces/:index | Read Trace events (paginated) | +| GET | /traces/:index/analysis | Trace performance analysis | +| GET | /scratchpad/:fileName | Read a session scratch file (e.g. input images) | + +General conventions: Sessions the user cannot access always return 404 — their existence is never leaked; only one Task or compaction runs per Session at a time, and conflicts return 409 (`task_in_progress` / `compacting`). + +Key request bodies (explicit keys): + +```ts +// POST /api/sessions/:sessionId/tasks — start a Task +interface TaskCreateRequest { + input: TaskInputPart[]; +} +type TaskInputPart = + | { type: "text"; text: string } + | { type: "image_url"; imageUrl: string }; // pasted images arrive as data URLs + +// POST /api/sessions/:sessionId/approvals/:toolCallId +interface ApprovalDecisionRequest { + decision: "allow" | "deny"; +} +``` + +## Streaming (SSE) + +Real-time delivery uses Server-Sent Events, not WebSocket, on two channels (the ordering semantics of what the channels carry are on [Message Flow & Ordering](/message-flow)): + +| Channel | Path | Contents | +| --- | --- | --- | +| Per Session | GET /api/sessions/:sessionId/stream | The Session's message stream and run events | +| Per user | GET /api/events | `hello` handshake and cross-Session notifications (schedule_fired / schedule_queued / session_created) | + +### Wire Format + +Default (unnamed) SSE events carry raw OmniMessage envelopes as single-line JSON — the same protocol the SDK yields and the Trace stores, see the [OmniMessage Protocol](/omni-message). Events named `server_event` carry the ServerEvent union: + +```ts +export type ServerEvent = + | { type: "approval_request"; toolCall: OmniMessage; origin?: string[] } + | { type: "task_state"; state: "idle" | "running" | "compacting" } + | { type: "session_title"; sessionId: string; title: string } + | { type: "resync_required" } + | { type: "hello" } + | { type: "session_created"; projectId: string; agentId: string; sessionId: string; source: SessionSource } + | { type: "schedule_fired"; projectId: string; agentId: string; name: string; sessionId: string } + | { type: "schedule_queued"; projectId: string; agentId: string; name: string; sessionId: string }; +``` + +| Event | Fired when | +| --- | --- | +| approval_request | A tool call escalated to human approval: every call under always-ask, plus rw / unknown-permission calls under read-only; pending approvals are resent on reconnect | +| task_state | The Session's run state flips (idle / running / compacting) | +| session_title | The model-generated title after the first turn has been persisted | +| resync_required | The Last-Event-ID was evicted from the buffer; the client must refetch history | +| hello | Handshake on the user channel | +| session_created | A new Session was registered (e.g. a subagent session) | +| schedule_fired | A scheduled task fired and was delivered | +| schedule_queued | The target Session is running; this firing was queued | + +### Delivery Guarantees + +- Event ids are monotonic per channel, shaped `-`; +- Each channel keeps a bounded replay buffer (most recent 1000 events or 2MB); +- Reconnecting with `Last-Event-ID` replays the gap on a buffer hit; on a miss the server first sends `resync_required`, and the client refetches `/messages` before continuing; +- A heartbeat comment line is written every 20 seconds; +- Event order: on a reconnect carrying `Last-Event-ID`, **the replayed gap (or `resync_required`) arrives first**, then the initial events — the authoritative `task_state` snapshot and still-pending approval_requests — then the live stream. A fresh connection (no `Last-Event-ID`) skips replay, so its first event is the `task_state` snapshot. + +### Recommended Client Pattern + +The order the bundled Web App uses: + +1. Connect `/stream` first and buffer incoming events; +2. GET `/messages` for the full history; +3. Replay the buffer, deduplicating the overlap; +4. Go live. + +## Type Imports + +All DTO types are importable type-only from the server package's `@prismshadow/penguin-server/api` subpath: + +```ts +import type { ServerEvent, SessionInfo } from "@prismshadow/penguin-server/api"; +``` diff --git a/packages/docs/content/server-api.zh.md b/packages/docs/content/server-api.zh.md new file mode 100644 index 0000000..1982c0f --- /dev/null +++ b/packages/docs/content/server-api.zh.md @@ -0,0 +1,236 @@ +--- +title: Server API +description: HTTP API 参考:认证机制、路由列表、SSE 流式协议与 DTO 类型导入。 +--- + +PenguinHarness Server 提供一套同源 HTTP API,自带的 Web App 与其他 HTTP 客户端都通过它访问。本文是接口参考:认证机制、路由列表与 SSE 流式协议。服务启动方式见[快速开始](/quickstart)。 + +## 总览 + +- 技术栈:Hono + @hono/node-server,要求 Node >= 24; +- 存储:SQLite(内置 `node:sqlite`,WAL 模式)仅存放索引与聚合数据——用户、登录会话、Project 授权、Agent / Session 索引、用量、UI 偏好、错误记录与 Schedule 状态;Agent、Trace 与 Workspace 数据全部以文件形式存放在 `~/.penguin/data` 下,与 CLI / SDK 共享,见[配置参考](/configuration); +- 监听:默认 `127.0.0.1:7364`,可用环境变量 `PORT` / `HOST` 调整; +- 请求体:写请求仅接受 JSON(Content-Type 校验,CSRF 防线之一),上限 20MB; +- 错误响应统一为: + +```text +{ "error": { "code": "<机器可读错误码>", "message": "<提示文案>" } } +``` + +## 目录结构 + +```text +packages/server/src +├── index.ts / config.ts / app.ts # 启动入口 · 环境变量配置 · Hono 组装(createApp 不绑端口,便于测试) +├── api/types.ts # 对外 DTO 契约(经 "./api" 子路径供前端 type-only 引用) +├── auth/ # scrypt 密码、admin 种子、cookie 会话、认证中间件 +├── db/ # node:sqlite 连接、建表 SQL、每表一个 repo +├── http/ # 错误体、请求校验、SSE 适配、routes/ 全部路由 +├── runtime/ # session-manager(运行时驱动)· channel(SSE 环形缓冲) +│ # approvals · usage-recorder · scheduler · title-generator +└── services/ # 授权规则、TOML/YAML 配置读写、Session/Trace/用量/快照服务 +``` + +## 认证 + +- Cookie 会话:`penguin_session`(HttpOnly、SameSite=Lax),有效期 7 天,滑动续期; +- 密码以 scrypt 哈希存储;服务端只保存会话 Token 的 sha256,不落明文; +- 不开放注册:启动时种子化内置管理员 `admin` / `admin123`,其余账号由管理员创建; +- 仅限同源访问,未启用 CORS 中间件。 + +```bash +curl -c cookies.txt -H "Content-Type: application/json" \ + -d '{"userId":"admin","password":"admin123"}' \ + http://127.0.0.1:7364/api/auth/login +``` + +## 路由参考 + +### 认证与账户 + +| 方法 | 路径 | 说明 | +| --- | --- | --- | +| POST | /api/auth/login | 登录:`{userId, password}` → `{user}` | +| POST | /api/auth/logout | 退出登录,返回 204 | +| GET | /api/me | 当前用户信息 | +| PUT | /api/me/password | 修改密码:`{oldPassword, newPassword}` | +| GET | /api/me/prefs | 读取 UI 偏好 | +| PUT | /api/me/prefs | 写入 UI 偏好(浅合并) | + +### 用户管理(仅管理员) + +| 方法 | 路径 | 说明 | +| --- | --- | --- | +| GET | /api/admin/users | 用户列表 | +| POST | /api/admin/users | 创建用户:`{userId, password}` | +| POST | /api/admin/users/:userId/password | 重置密码(该用户全部登录会话失效) | +| DELETE | /api/admin/users/:userId | 删除用户 | + +### Project 与成员 + +| 方法 | 路径 | 说明 | +| --- | --- | --- | +| GET | /api/projects | 当前用户可见的 Project 列表 | +| POST | /api/projects | 创建 Project | +| DELETE | /api/projects/:projectId | 删除 Project | +| GET | /api/projects/:projectId/members | 成员列表 | +| POST | /api/projects/:projectId/members | 添加成员:`{userId}` | +| DELETE | /api/projects/:projectId/members/:userId | 移除成员 | + +成员写操作仅限 Owner。 + +### 模型 + +| 方法 | 路径 | 说明 | +| --- | --- | --- | +| GET | /api/projects/:projectId/models | 模型列表(api_key 掩码显示) | +| PUT | /api/projects/:projectId/models | 全表替换,条目以 `(provider, modelId)` 为键 | +| POST | /api/projects/:projectId/models/test | 连通性测试:`{provider, modelId, …}` → `{ok, latencyMs?, message?}` | + +### Agent + +以下路径均省略前缀 `/api/projects/:projectId`。 + +| 方法 | 路径 | 说明 | +| --- | --- | --- | +| GET / POST | /agents | Agent 列表 / 创建 | +| DELETE | /agents/:agentId | 删除 Agent | +| GET / PUT | /agents/:agentId/config | 读写配置(AGENTS.md + system_config.yaml,PUT 保留 YAML 注释) | +| GET / PUT | /agents/:agentId/vault | Vault 环境变量(值掩码显示;PUT 全表替换) | +| GET | /agents/:agentId/export | 导出 Agent State 快照(tar.gz 下载) | +| POST | /agents/:agentId/import | 导入快照:`{dataBase64, confirm?}`;版本冲突且未确认时返回 409 | +| GET / POST | /agents/:agentId/skills | 已安装 Skill 列表 / 安装 | +| DELETE | /agents/:agentId/skills/:name | 卸载 Skill | +| GET | /agents/:agentId/benchmarks | Benchmark 评分数据(只读) | + +### Schedule + +| 方法 | 路径 | 说明 | +| --- | --- | --- | +| GET / POST | /agents/:agentId/schedules | 定时任务列表 / 创建(重名返回 409) | +| GET / PUT / DELETE | /agents/:agentId/schedules/:name | 读取 / 更新 / 删除单个任务 | + +Schedule 写操作仅限 Owner。 + +### Session 创建与目录浏览 + +| 方法 | 路径 | 说明 | +| --- | --- | --- | +| GET | /agents/:agentId/sessions | Session 列表(含运行状态) | +| POST | /agents/:agentId/sessions | 创建 Session:`{modelId?, provider?, workspace?, approvalMode?}` → 201 | +| GET | /dirs?path= | 服务器端目录浏览(Workspace 选择器数据源) | + +创建 Session 时,模型默认取 Project 默认模型,Workspace 默认自动创建临时目录,审批模式默认 `allow-all`。 + +### 用量与 Trace(Agent 级) + +| 方法 | 路径 | 说明 | +| --- | --- | --- | +| GET | /usage | 用量统计,查询参数 `from`、`to`、`groupBy`、`agentId`、`provider`、`modelId` | +| GET | /agents/:agentId/traces | Trace 文件的日期 → Session 下钻结构 | +| GET | /agents/:agentId/traces/:sessionId/:index | 读取 Trace 事件(`offset` / `limit` 分页) | +| GET | /agents/:agentId/traces/:sessionId/:index/analysis | Trace 性能分析结果 | + +### Session 级接口 + +以下路径均省略前缀 `/api/sessions/:sessionId`。Trace 与 Session 的存储模型见 [Session 与 Trace](/sessions-and-traces)。 + +| 方法 | 路径 | 说明 | +| --- | --- | --- | +| GET | / | Session 信息 | +| PATCH | / | 更新:`{approvalMode?, archived?, title?}` | +| DELETE | / | 删除 Session(连同 Trace 与暂存文件) | +| GET | /messages | 完整 OmniMessage 历史 | +| GET | /stream | SSE 事件流(见下节) | +| POST | /tasks | 发起 Task:`{input: TaskInputPart[]}` → 202 | +| POST | /approvals/:toolCallId | 审批决定:`{decision}` 取 `allow` 或 `deny` → 204 | +| POST | /abort | 中断当前 Task:已触发返回 202,无任务返回 204 | +| POST | /compact | 触发上下文压缩:202;无可压缩内容返回 409 `nothing_to_compact` | +| GET | /files?path= | 浏览 Workspace 目录 | +| GET | /files/content?path=&download= | 读取 Workspace 文件(`download=1` 时作为附件下载) | +| POST | /files/stat | 批量存在性检查:`{paths}` | +| PUT | /files/content?path= | 上传文件:`{dataBase64}`,上限 14MB | +| GET | /traces | 本 Session 的 Trace 文件列表 | +| GET | /traces/:index | 读取 Trace 事件(分页) | +| GET | /traces/:index/analysis | Trace 性能分析结果 | +| GET | /scratchpad/:fileName | 读取会话暂存文件(如输入图片) | + +通用约定:无权访问的 Session 一律返回 404,不泄露其存在性;每个 Session 同时只允许一个 Task 或压缩在运行,冲突时返回 409(`task_in_progress` / `compacting`)。 + +关键请求体(明确键名): + +```ts +// POST /api/sessions/:sessionId/tasks —— 发起一个 Task +interface TaskCreateRequest { + input: TaskInputPart[]; +} +type TaskInputPart = + | { type: "text"; text: string } + | { type: "image_url"; imageUrl: string }; // 粘贴图片以 data URL 上送 + +// POST /api/sessions/:sessionId/approvals/:toolCallId +interface ApprovalDecisionRequest { + decision: "allow" | "deny"; +} +``` + +## 流式接口(SSE) + +实时通道采用 Server-Sent Events 而非 WebSocket,共两条(通道内承载的消息顺序语义见[消息流转与时序](/message-flow)): + +| 通道 | 路径 | 内容 | +| --- | --- | --- | +| Session 级 | GET /api/sessions/:sessionId/stream | 该 Session 的消息流与运行事件 | +| 用户级 | GET /api/events | `hello` 握手与跨 Session 通知(schedule_fired / schedule_queued / session_created) | + +### 传输格式 + +默认(未命名)SSE 事件承载原始 OmniMessage 信封(单行 JSON)——与 SDK 产出、Trace 落盘是同一套协议,见 [OmniMessage 协议](/omni-message);命名为 `server_event` 的事件承载 ServerEvent 联合类型: + +```ts +export type ServerEvent = + | { type: "approval_request"; toolCall: OmniMessage; origin?: string[] } + | { type: "task_state"; state: "idle" | "running" | "compacting" } + | { type: "session_title"; sessionId: string; title: string } + | { type: "resync_required" } + | { type: "hello" } + | { type: "session_created"; projectId: string; agentId: string; sessionId: string; source: SessionSource } + | { type: "schedule_fired"; projectId: string; agentId: string; name: string; sessionId: string } + | { type: "schedule_queued"; projectId: string; agentId: string; name: string; sessionId: string }; +``` + +| 事件 | 触发时机 | +| --- | --- | +| approval_request | 工具调用升级为人工审批时发出:always-ask 下的所有调用,以及 read-only 下 rw / 未知权限的调用;重连时未决审批会重发 | +| task_state | Session 运行状态翻转(idle / running / compacting) | +| session_title | 首轮后模型生成的标题已持久化 | +| resync_required | Last-Event-ID 已被缓冲区淘汰,客户端须重新拉取历史 | +| hello | 用户通道连接握手 | +| session_created | 新 Session 注册(如子 Agent 会话) | +| schedule_fired | 定时任务已触发并发送 | +| schedule_queued | 目标 Session 正在运行,本次触发已排队 | + +### 投递保证 + +- 事件 id 按通道单调递增,形如 `-`; +- 每通道维护有界重放缓冲(最近 1000 条事件或 2MB); +- 携带 `Last-Event-ID` 重连时,命中缓冲则补发缺口;未命中则先发 `resync_required`,客户端重新拉取 `/messages` 后继续消费; +- 每 20 秒写一条心跳注释行; +- 事件次序:带 `Last-Event-ID` 重连时,**补发的缺口(或 `resync_required`)最先送达**,随后才是初始事件——权威的 `task_state` 快照与未决的 approval_request,再进入实时流;全新连接(无 `Last-Event-ID`)不重放缓冲,首个事件即为 `task_state` 快照。 + +### 推荐客户端模式 + +自带 Web App 的接入顺序: + +1. 先连接 `/stream` 并缓冲收到的事件; +2. 再 GET `/messages` 拉取完整历史; +3. 回放缓冲区并对重叠消息去重; +4. 转入实时消费。 + +## 类型导入 + +全部 DTO 类型可从服务端包的子路径 `@prismshadow/penguin-server/api` 以 type-only 方式导入: + +```ts +import type { ServerEvent, SessionInfo } from "@prismshadow/penguin-server/api"; +``` diff --git a/packages/docs/content/sessions-and-traces.en.md b/packages/docs/content/sessions-and-traces.en.md new file mode 100644 index 0000000..a8fad63 --- /dev/null +++ b/packages/docs/content/sessions-and-traces.en.md @@ -0,0 +1,88 @@ +--- +title: Sessions & Traces +description: The six-level run model, local data directory layout, Trace file design, and Session recovery. +--- + +All PenguinHarness runtime data lives on the local file system: configuration is editable files, history is append-only Traces. This page defines each level of the run model and explains how the Trace serves as history, recovery source, and statistics source at once. + +## Run model + +Six levels: Project → Agent → Workspace → Session → Task → Request. + +| Concept | Definition | +| --- | --- | +| Project | Top-level unit organizing Agents; owns the model and credential configuration; in the multi-user Web setup, users and Projects are many-to-many | +| Agent | The executing subject; has exactly one Agent State (a persistent directory); one Agent can serve many Workspaces | +| Workspace | The working directory of one run — the only file scope the model sees; an explicit `workspaceDir` must already exist, otherwise a temp Workspace `workspaces/tmp-<8hex>` is created | +| Session | A continuous conversation under one (Agent, Workspace); model and Workspace are locked at Session creation; ids look like `session-YYYY-MM-DD-HH-mm-ss-<8hex>` | +| Task | One execution goal started by one Prompt; consists of one or more consecutive Requests | +| Request | One LLM API call: context and tool definitions in, streamed output out | + +See the [Architecture](/architecture) page for how the levels cooperate, and the [Agent Loop](/agent-loop) for how Requests advance within a Task. + +## Data layout + +The data root is the `PENGUIN_HOME` environment variable, defaulting to `~/.penguin/data`. The layout is defined in one place, `packages/core/src/state/paths.ts`: + +```text +// +├── .project_config.toml # Project-level models & credentials (hidden file, 0600) +└── agents/ + └── / + ├── agent_state/ # system_config.yaml, AGENTS.md, .vault.toml, + │ # tools/, memory/, skills/, schedule/ + ├── traces/ + │ └── /_.jsonl + ├── scratchpad/ # temp files, one subdirectory per Session id (e.g. pasted images) + ├── workspaces/ # temp Workspaces (tmp-<8hex>) + ├── benchmarks/ # capability Benchmark cases and scores + └── snapshots/ # Agent State version snapshots +``` + +See the [Configuration Reference](/configuration) for the fields of each config file. + +## Trace design + +A Trace is an append-only JSON Lines file; each line is one OmniMessage envelope (see the [OmniMessage Protocol](/omni-message)). History is only ever appended, never modified in place. + +- One Trace file corresponds to one complete model context. When compaction produces a new context segment, the writer rotates to a new file — `_002`, `_003`, … — with an incrementing index. +- Recorded: `session_meta`, complete `model_msg`, and all `event_msg`. +- Not recorded: streaming `partial_*` fragments (the producer appends the complete message once the segment ends), and nested messages tagged with `origin` — a subagent's messages go to the child Session's own Trace, while the parent Trace keeps a single `subagent` pointer event at the spawn site recording the child Session id. +- `request_begin` and `request_end(status)` come in pairs delimiting one Request; replay uses `request_end.status === "completed"` as the commit criterion for that turn. + +See `packages/core/src/trace/writer.ts` for the implementation. + +The head of a Trace (illustrative; one OmniMessage envelope per line): + +```jsonl +{"timestamp":"2026-07-18T03:10:22.531Z","type":"session_meta","payload":{"session_id":"session-2026-07-18-11-10-22-3f8a1c2d","provider":"deepseek","model_id":"deepseek-v4-pro","model_context_window":1000000,"system_prompt":"…","tools":[…],"thinking_level":"medium","agent_state":"/home/u/.penguin/data/default_project/agents/default_agent/agent_state","workspace":"/home/u/work"}} +{"timestamp":"…","type":"event_msg","payload":{"type":"request_begin"}} +{"timestamp":"…","type":"model_msg","payload":{"type":"text","role":"user","text":"Create hello.txt"}} +{"timestamp":"…","type":"model_msg","payload":{"type":"tool_call","role":"assistant","name":"exec_command","arguments":"{\"cmd\":\"printf hi > hello.txt\"}","tool_call_id":"call_0"}} +{"timestamp":"…","type":"event_msg","payload":{"type":"approval_decision","decision":"allow","tool_call_id":"call_0"}} +{"timestamp":"…","type":"model_msg","payload":{"type":"tool_call_output","role":"user","output":"[no output]","tool_call_id":"call_0"}} +{"timestamp":"…","type":"event_msg","payload":{"type":"request_end","status":"completed"}} +{"timestamp":"…","type":"event_msg","payload":{"type":"token_usage","session":{…},"request":{…}}} +``` + +## Session recovery + +The Trace is the single source of truth for recovery — there is no separate session database to keep in sync. `resumeSession` works as follows: + +1. Locate the highest-index Trace file of the Session; +2. Read the runtime configuration from its `session_meta` — model, system prompt, Workspace — all immutable for the lifetime of the Session; +3. Replay the committed history into a fresh LLM context; +4. Reconstruct the carry-over (undelivered tool outputs, interruption markers) plus turn and Token counters; +5. Continue appending to the same Trace file. + +Recovery requires that the Workspace and the model still exist. What recovery guarantees is structural legality: only committed turns are replayed, with `tool_call` / `tool_call_output` pairing intact; incomplete model output (thinking, text) is allowed to be lost. A truncated last line left by an abnormal process exit is tolerated and ignored. See `packages/core/src/trace/resume.ts`. + +Special case: if the latest Trace file ends with a completed compaction, that context is closed as a whole — resume starts from an empty context; in summarize mode the `` is reconstructed and prepended to the first input after resume. + +## Field fidelity + +Provider-specific fields (such as `signature` and `phase`) are preserved verbatim in the Trace and sent back verbatim — some models require them byte-for-byte on history replay, and any rewriting would break compatibility. This is one reason the Trace stores raw OmniMessage envelopes rather than a post-processed format. + +## Observability + +Every approval decision (`approval_decision`), abort (`abort`), compaction (`compaction_begin` / `compaction_end`), and `token_usage` lands in the Trace as an event. The Web Trace view and the usage/cost statistics are both derived from this same data — there is no second source of truth; see the [Web App Guide](/web-app). The approval mechanism itself is covered in [Tools & Approval](/tools). diff --git a/packages/docs/content/sessions-and-traces.zh.md b/packages/docs/content/sessions-and-traces.zh.md new file mode 100644 index 0000000..8890185 --- /dev/null +++ b/packages/docs/content/sessions-and-traces.zh.md @@ -0,0 +1,88 @@ +--- +title: Session 与 Trace +description: 运行模型的六层结构、本地数据目录布局、Trace 文件设计与 Session 恢复机制。 +--- + +PenguinHarness 的全部运行数据都落在本地文件系统:配置是可编辑文件,历史是 append-only 的 Trace。本页定义运行模型的各层概念,并说明 Trace 如何同时充当历史记录、恢复依据与统计来源。 + +## 运行模型 + +六层结构:Project → Agent → Workspace → Session → Task → Request。 + +| 概念 | 定义 | +| --- | --- | +| Project | 组织 Agent 的顶层单位,持有模型与凭证配置;Web 多用户部署中,用户与 Project 是多对多关系 | +| Agent | 执行主体,恰好拥有一份 Agent State(持久化目录);一个 Agent 可服务多个 Workspace | +| Workspace | 一次运行的工作目录,是模型可见的唯一文件范围;显式指定的 `workspaceDir` 必须已存在,未指定时自动创建临时 Workspace `workspaces/tmp-<8hex>` | +| Session | 同一(Agent、Workspace)下的一段连续对话;模型与 Workspace 在 Session 创建时锁定,id 形如 `session-YYYY-MM-DD-HH-mm-ss-<8hex>` | +| Task | 由一条 Prompt 发起的一个执行目标,由一个或多个连续的 Request 组成 | +| Request | 一次 LLM API 调用:上下文与工具定义送入,流式输出返回 | + +各层组件如何协作见[架构总览](/architecture);Task 内部 Request 如何推进见 [Agent 运行循环](/agent-loop)。 + +## 数据目录 + +数据根目录取环境变量 `PENGUIN_HOME`,缺省 `~/.penguin/data`。目录布局由 `packages/core/src/state/paths.ts` 统一定义: + +```text +// +├── .project_config.toml # Project 级模型与凭证(隐藏文件,0600) +└── agents/ + └── / + ├── agent_state/ # system_config.yaml、AGENTS.md、.vault.toml、 + │ # tools/、memory/、skills/、schedule/ + ├── traces/ + │ └── /_.jsonl + ├── scratchpad/ # 临时文件,按 Session id 建子目录(如粘贴的图片) + ├── workspaces/ # 临时 Workspace(tmp-<8hex>) + ├── benchmarks/ # 能力评测题库与得分 + └── snapshots/ # Agent State 版本快照 +``` + +配置文件的字段详见[配置参考](/configuration)。 + +## Trace 设计 + +Trace 是 append-only 的 JSON Lines 文件,每行一个 OmniMessage 信封(协议见 [OmniMessage 协议](/omni-message))。历史事件只追加、从不原地修改。 + +- 一个 Trace 文件对应一份完整的模型上下文。上下文压缩产生新的上下文段时,写入器轮转到 `_002`、`_003`……新文件,索引递增。 +- 记录的消息:`session_meta`、完整的 `model_msg`、全部 `event_msg`。 +- 不记录的消息:流式 `partial_*` 分片(片段结束后由生产方补写完整消息);带 `origin` 标记的嵌套消息——子 Agent 的消息写入子 Session 自己的 Trace,父 Trace 只在派生位置保留一个 `subagent` 指针事件,记录子 Session id。 +- `request_begin` 与 `request_end(status)` 成对出现,界定一轮 Request;回放以 `request_end.status === "completed"` 作为该轮已提交的判据。 + +实现见 `packages/core/src/trace/writer.ts`。 + +一条 Trace 的开头(示意,每行一个 OmniMessage 信封): + +```jsonl +{"timestamp":"2026-07-18T03:10:22.531Z","type":"session_meta","payload":{"session_id":"session-2026-07-18-11-10-22-3f8a1c2d","provider":"deepseek","model_id":"deepseek-v4-pro","model_context_window":1000000,"system_prompt":"…","tools":[…],"thinking_level":"medium","agent_state":"/home/u/.penguin/data/default_project/agents/default_agent/agent_state","workspace":"/home/u/work"}} +{"timestamp":"…","type":"event_msg","payload":{"type":"request_begin"}} +{"timestamp":"…","type":"model_msg","payload":{"type":"text","role":"user","text":"创建 hello.txt"}} +{"timestamp":"…","type":"model_msg","payload":{"type":"tool_call","role":"assistant","name":"exec_command","arguments":"{\"cmd\":\"printf hi > hello.txt\"}","tool_call_id":"call_0"}} +{"timestamp":"…","type":"event_msg","payload":{"type":"approval_decision","decision":"allow","tool_call_id":"call_0"}} +{"timestamp":"…","type":"model_msg","payload":{"type":"tool_call_output","role":"user","output":"[no output]","tool_call_id":"call_0"}} +{"timestamp":"…","type":"event_msg","payload":{"type":"request_end","status":"completed"}} +{"timestamp":"…","type":"event_msg","payload":{"type":"token_usage","session":{…},"request":{…}}} +``` + +## Session 恢复 + +Trace 是恢复的唯一事实来源,没有独立的会话数据库需要与之对齐。`resumeSession` 的流程: + +1. 定位该 Session 索引最大的 Trace 文件; +2. 从文件内的 `session_meta` 读取运行配置——模型、系统提示词、Workspace,三者在 Session 生命周期内不可变; +3. 将已提交的历史回放进一份全新的 LLM 上下文; +4. 重建 carry-over(未送达的工具输出、中断标记)与轮数、Token 计数器; +5. 继续追加写入同一个 Trace 文件。 + +恢复的前提是 Workspace 与模型仍然存在。恢复保证的是结构合法性:只回放已提交的轮次,`tool_call` 与 `tool_call_output` 配对完整;未完成的模型输出(thinking、文本)允许丢失。异常退出留下的截断末行会被容忍并忽略。实现见 `packages/core/src/trace/resume.ts`。 + +特殊情形:若最新 Trace 文件以一次完成的压缩收尾,则该上下文已整体关闭——恢复从空上下文开始;summarize 模式下会重建 `` 摘要,前置到恢复后第一轮输入中。 + +## 字段保真 + +Provider 专有字段(如 `signature`、`phase`)在 Trace 中原样保存、原样回传——部分模型在历史回放时要求这些字段逐字一致,任何转写都会破坏兼容性。这也是 Trace 直接存储 OmniMessage 信封而非二次加工格式的原因之一。 + +## 可观测性 + +每一次审批决策(`approval_decision`)、中断(`abort`)、压缩(`compaction_begin` / `compaction_end`)与 `token_usage` 都作为事件落入 Trace。Web 的 Trace 视图与用量、成本统计均由这同一份数据派生,不存在第二事实来源;见 [Web App 指南](/web-app)。审批机制本身见[工具与审批](/tools)。 diff --git a/packages/docs/content/skills.en.md b/packages/docs/content/skills.en.md new file mode 100644 index 0000000..517debf --- /dev/null +++ b/packages/docs/content/skills.en.md @@ -0,0 +1,76 @@ +--- +title: Skills +description: Skills package reusable instructions as directories with a SKILL.md — metadata up front, body on demand, editable by the Agent itself. +--- + +## Anatomy of a Skill + +A Skill is a directory containing a `SKILL.md`, optionally with a custom `icon.svg`. The directory name is the authoritative skill name and must match `^[A-Za-z0-9_-]+$`; a `name` in the frontmatter is overridden by it. + +Frontmatter fields: + +| Field | Meaning | +| --- | --- | +| `name` | Skill name, matching the directory name | +| `description` | English one-liner injected into the system prompt | +| `short_description` / `short_description_zh` | UI labels for compact spots such as cards; not injected into the prompt | +| `version` | Natural-number version, default 1 | +| `updated` | Update date | + +```md +--- +name: my-skill +description: One-line English description injected into the system prompt. +short_description: Short UI label. +short_description_zh: 简短的中文标签。 +version: 1 +updated: 2026-07-17 +--- + +# My Skill + +Concrete steps, boundaries and acceptance criteria... +``` + +Parsing is tolerant: only `key: value` scalar lines inside the first `---` block are recognized; a `version` that is not a natural number falls back to 1, and a missing `updated` defaults to empty. + +## Progressive loading + +Skills follow an "index first, body on demand" design: the system prompt injects only each installed Skill's metadata (name + description) through the `{{SKILL_METADATA}}` placeholder, and instructs the model to read the matching `SKILL.md` in full via the shell before following it. There is no dedicated skill tool — reading the body is just one `exec_command` call (see [Tools & Approval](/tools)). + +Chat can also pin skills explicitly: the message then starts with a `` block listing the skill names. + +If a message only names a skill without a concrete task, the model is instructed to ask what is needed before starting. + +## Installation and storage + +Installed Skills live under `agent_state/skills//` inside the Agent State. The files are the source of truth: every read goes straight to disk with no cache, which makes Skills naturally editable. + +- The built-in Agent `default_agent` gets the whole library installed at initialization; +- other Agents install on demand — through the Web UI's Skill library page, or via the SDK; +- installing writes the library `SKILL.md` verbatim (frontmatter included) and copies any `icon.svg` alongside it. + +The library ships as the npm package `@prismshadow/penguin-skills`, carrying the raw `skills/` directory in the tarball; at runtime the package's `skills//SKILL.md` files are likewise the source of truth for library content. + +## Built-in library + +The built-in Skills, by group (the group manifest is `SKILL_GROUPS` in `packages/skills/src/index.ts`; the library directory is the source of truth as Skills are added): + +| Group | Skill | Purpose | +| --- | --- | --- | +| Agent Development | `agent-creation` | Turn a user requirement into a concrete agent: write the target agent's AGENTS.md and install the skills it needs | +| | `benchmark-design` | Design and calibrate a multi-Case capability Benchmark with repeated independent evaluations and a traceable baseline | +| | `agent-evaluation` | Run and score exactly one Benchmark Case run, with CLI execution, Trace provenance checks and private Rubric isolation | +| | `agent-optimization` | Improve an Agent State from direct feedback or versioned multi-Case Benchmark scores and score-linked Traces | +| Data Analysis | `data-analysis` | Complete data-analysis tasks with bounded evidence inspection, explicit answer-changing decisions, native artifact handling and final output verification | +| Penguin Development | `penguin-sdk` | Build AI apps on the SDK (the createSession/run streaming loop) | +| | `penguin-cli` | Manage model API keys, default models and per-agent Vault secrets with the penguin CLI | +| | `agenthub-models` | Call model APIs through `@prismshadow/agenthub`: streaming text, image generation, speech synthesis and embeddings | +| Web Development | `web-design` | Default visual language for generated web pages: minimal black-white-gray | +| Software Engineering | `software-engineering` | Complete software-engineering tasks: investigate and review code, implement fixes, features and refactors with minimal scope, validate changes, and report verified outcomes | + +## Writing and optimizing Skills + +- Manual install: create a directory under `agent_state/skills//` and write a `SKILL.md`; the system scans `skills/` when assembling the system prompt and injects the metadata. A directory without a `SKILL.md` does not count as a Skill. +- Uninstalling deletes the whole `skills//` directory and is idempotent. +- An Agent can rewrite its own SKILL.md as part of a task — combined with Benchmark evaluation and optimization this closes the improvement loop, see [Self-Improvement](/self-improvement). diff --git a/packages/docs/content/skills.zh.md b/packages/docs/content/skills.zh.md new file mode 100644 index 0000000..e630604 --- /dev/null +++ b/packages/docs/content/skills.zh.md @@ -0,0 +1,76 @@ +--- +title: 技能系统 +description: Skill 以目录加 SKILL.md 承载可复用指令,元数据先行、正文按需读取,并可由 Agent 自行编辑优化。 +--- + +## Skill 的形态 + +一个 Skill 就是一个目录:内含一份 `SKILL.md`,可选附带一个 `icon.svg` 自定义图标。目录名即权威的 Skill 名,须匹配 `^[A-Za-z0-9_-]+$`;frontmatter 中的 `name` 以目录名为准。 + +frontmatter 字段: + +| 字段 | 说明 | +| --- | --- | +| `name` | Skill 名,与目录名一致 | +| `description` | 英文单行描述,注入系统 Prompt | +| `short_description` / `short_description_zh` | UI 短标签(卡片等紧凑位置用),不注入 Prompt | +| `version` | 自然数版本号,默认 1 | +| `updated` | 更新日期 | + +```md +--- +name: my-skill +description: One-line English description injected into the system prompt. +short_description: Short UI label. +short_description_zh: 简短的中文标签。 +version: 1 +updated: 2026-07-17 +--- + +# My Skill + +具体的步骤、边界与验收标准…… +``` + +解析是容错的:只识别首个 `---` 块内的 `key: value` 标量行;`version` 不是自然数时回退为 1,`updated` 缺省为空。 + +## 渐进式加载 + +Skill 采用「先索引、后正文」的设计:系统 Prompt 经 `{{SKILL_METADATA}}` 占位符只注入每个已安装 Skill 的元数据(name + description),并指示模型在任务匹配某个 Skill 时,先用 Shell 完整读取对应的 `SKILL.md`,再遵循执行。系统不设专门的 Skill 工具,读取正文就是一次 `exec_command` 调用(见 [工具与审批](/tools))。 + +对话中也可以显式指定 Skill:此时消息以 `` 块开头,列出要使用的 Skill 名。 + +若消息只点名 Skill 而没有给出具体任务,模型会先询问需求再开始。 + +## 安装与存放 + +已安装的 Skill 位于 Agent State 的 `agent_state/skills//`。文件即事实源:每次读取直接读文件、不设缓存,因此 Skill 天然可编辑。 + +- 内置 Agent `default_agent` 在初始化时安装完整 Skill 库; +- 其他 Agent 按需安装:经 Web 界面的 Skill 库页,或经 SDK; +- 安装即把库里的 `SKILL.md` 原样写入(含 frontmatter),目录内的 `icon.svg` 一并拷贝。 + +Skill 库以 npm 包 `@prismshadow/penguin-skills` 发布,tarball 直接携带原始 `skills/` 目录;运行时库内容的事实源同样是包内的 `skills//SKILL.md` 文件。 + +## 内置 Skill 库 + +内置 Skill 按分组列出如下(分组清单见 `packages/skills/src/index.ts` 的 `SKILL_GROUPS`,新增 Skill 时以库目录为准): + +| 分组 | Skill | 说明 | +| --- | --- | --- | +| Agent 开发 | `agent-creation` | 把用户需求变成具体的 Agent:撰写目标 Agent 的 AGENTS.md 并安装所需 Skill | +| | `benchmark-design` | 设计并校准多 Case 的能力评测 Benchmark,含重复独立评测与可追溯基线 | +| | `agent-evaluation` | 隔离执行并评分单个 Benchmark Case:CLI 执行、Trace 溯源检查、Rubric 私有隔离 | +| | `agent-optimization` | 依据直接反馈或带版本的多 Case Benchmark 分数与关联 Trace 改进 Agent State | +| 数据分析 | `data-analysis` | 以有界的证据检查、显式的改答案决策、原生产物处理与最终输出校验完成数据分析任务 | +| Penguin 开发 | `penguin-sdk` | 基于 SDK 构建 AI 应用(createSession/run 流式循环) | +| | `penguin-cli` | 用 penguin CLI 管理模型 API Key、默认模型与各 Agent 的 Vault 密钥 | +| | `agenthub-models` | 经 `@prismshadow/agenthub` 调用模型 API:流式文本、图像生成、语音合成与 Embedding | +| 网页开发 | `web-design` | 生成网页的默认视觉规范:极简黑白灰 | +| 软件工程 | `software-engineering` | 完成软件工程任务:调查与审查代码,以最小改动实现修复、特性与重构,验证改动并报告经过确认的结果 | + +## 编写与优化 + +- 手工安装:在 `agent_state/skills//` 下建目录并写入 `SKILL.md` 即可,系统组装系统 Prompt 时扫描 `skills/` 注入元数据;没有 `SKILL.md` 的目录不计为 Skill。 +- 卸载即删除整个 `skills//` 目录,操作幂等。 +- Agent 可以在任务中直接改写自己的 SKILL.md——配合 Benchmark 评测与优化形成闭环,见 [自我进化](/self-improvement)。 diff --git a/packages/docs/content/tools.en.md b/packages/docs/content/tools.en.md new file mode 100644 index 0000000..b9816bd --- /dev/null +++ b/packages/docs/content/tools.en.md @@ -0,0 +1,189 @@ +--- +title: Tools & Approval +description: The deliberately minimal built-in toolset, its execution contract with centralized close-out, and per-call approval audited in the Trace. +--- + +## Design + +PenguinHarness ships a deliberately minimal built-in toolset: the shell is the universal interface, and reading, writing and editing files all go through `exec_command` — there are no separate file tools. Fewer tools mean fewer schema tokens and fewer wrong calls. + +## Execution contract + +Every built-in tool implements the same `BuiltinTool` interface (`packages/core/src/environment/tools/types.ts`): + +```ts +interface BuiltinTool { + name: string; + definition: ToolDefinitionConfig; + execute( + args: Record, + ctx: ToolExecutionContext, + ): AsyncGenerator; +} + +interface ToolExecutionContext { + workspaceDir: string; + toolCallId: string; + signal?: AbortSignal; + approve?: ApproveFn; // forwarded to tools that spawn child Sessions (approval inheritance) +} + +interface ToolResult { + stopReason?: StopReason; // the tool's self-reported terminal state (lowest priority, see below) + note?: string; // terminal marker appended outside the truncation window (e.g. exit code) + images?: string[]; // data-URL images, appended after the text output +} +``` + +A tool only yields incremental `partial_tool_call_output` deltas; the Environment handles the close-out centrally: + +- streaming framing (start / stop) and `tool_call_id` threading; +- timeout merging, and head-kept truncation once output exceeds `maxOutputLength` (default 16000 characters); +- stop_reason priority: user interrupt > timeout > tool throw > tool self-report; +- never-empty output (`[no output]` is substituted when a tool produced nothing); +- `note` (e.g. the exit code) and images are appended outside the truncation window, so the terminal marker survives even when long output is cut. + +Tools and the Environment never throw into the engine: errors collapse into `tool_call_output` messages the model can read and react to. See the [OmniMessage Protocol](/omni-message) for message structure. + +## Configuration fields + +Each tool is described by one `ToolDefinitionConfig`: + +| Field | Meaning | +| --- | --- | +| `name` | Tool name, matching the model's `tool_call.name` | +| `description` | Tool description handed to the model | +| `parameters` | JSON Schema of the arguments | +| `permission` | `"r"` read-only / `"rw"` read-write | +| `forModel` | `"vision"` / `"text-only"`: selected by the Session model's class; omitted = available to all models | +| `timeoutMs` | Per-call timeout (ms), default 120000; `<=0` disables | +| `maxOutputLength` | Output length cap (characters); `<=0` disables | + +## Built-in tools + +There are 6 built-in tools (assembled via `packages/core/src/environment/tools/registry.ts`): + +| Tool | Permission | Timeout (ms) | Purpose | +| --- | --- | --- | --- | +| `exec_command` | rw | 120000 | Run a shell command in the Workspace via `bash -lc`, streaming stdout/stderr | +| `input_command` | rw | 130000 | Drive a running command by `process_id`: write stdin, send Ctrl-C, poll output | +| `run_subagent` | rw | 600000 | Delegate a self-contained subtask to a child Agent in the same Workspace | +| `input_subagent` | rw | 600000 | Poll a background subagent, or send a follow-up prompt once it is idle | +| `read_image` | r | 60000 | Read an image and return it as image content (vision models) | +| `describe_image` | r | 90000 | Have the configured `vision_model` read the image and answer in text (text-only models) | + +### Command sessions + +`exec_command` waits in the foreground first; if the command outruns `yield_time_ms` it moves to the background and the call returns the output so far plus a `process_id`, driven from then on by `input_command`: + +```text +exec_command(cmd) + ├─ finishes within the foreground window (yield_time_ms, default 60000) + │ ──► full output + exit code + └─ still running ──► backgrounds, returns output so far + process_id + │ + input_command(process_id[, chars]) ──► write stdin / send Ctrl-C / poll + └─ loop until the command exits +``` + +Both tools' arguments (explicit keys): + +```ts +// exec_command +{ + cmd: string; // required: the shell command to run + workdir?: string; // working directory; defaults to the Workspace root, relative paths resolve against it + yield_time_ms?: number; // foreground wait; default 60000, minimum 250, capped below the tool timeout +} + +// input_command +{ + process_id: string; // required: the command-session id returned by exec_command + chars?: string; // characters for stdin; send "\u0003" alone to deliver Ctrl-C; empty = poll only + yield_time_ms?: number; // wait; defaults 250 for writes, 5000 for empty polls +} +``` + +### Subagents + +`run_subagent` hands a subtask you can fully specify in one prompt to a child Agent, with the same two-phase shape: after the foreground window (default 300000ms) it moves to the background with a `subagent_id`, driven by `input_subagent` for polling or follow-up prompts; the child's pending approvals surface while the poll waits. + +```ts +// run_subagent +{ + prompt: string; // required: the complete subtask (all context + the exact final output expected) + agent_id?: string; // the child Agent; defaults to the current Agent + model_id?: string; // the child Session's model; defaults to the Project default model + yield_time_ms?: number; // foreground wait; default 300000 +} + +// input_subagent +{ + subagent_id: string; // required: the background Subagent id returned by run_subagent + prompt?: string; // follow-up task, accepted only while the child Session is idle; empty = poll only + yield_time_ms?: number; // wait; defaults 300000 with a prompt, 10000 for empty polls +} +``` + +- Depth is capped at 1: a subagent cannot spawn another subagent. +- The child Session inherits the parent Agent's approval callback, so the approval mode follows the parent. +- The child Session gets its own Trace, linked from the parent by a `subagent` pointer event; child messages stream back into the parent flow tagged with `origin`. See [Sessions & Traces](/sessions-and-traces). + +### Image tools + +`read_image` and `describe_image` are mutually exclusive, selected by the Session model's vision flag. Both accept an http(s) URL or a Workspace path and support png/jpeg/gif/webp up to 5MB. Text-only models get `describe_image`: the image plus a prompt are forwarded to the Project's configured `vision_model`, whose text answer becomes the tool output. See [Models & Providers](/models). + +```ts +// read_image (vision models) +{ + source: string; // required: an http(s) URL, or a file path inside the Workspace +} + +// describe_image (text-only models) +{ + source: string; // required: as above + prompt?: string; // what to ask about the image; defaults to a detailed description +} +``` + +### Background session caps + +| Session type | Cap | Eviction | +| --- | --- | --- | +| Command sessions | 64 | When full, exited sessions are evicted first, then idle ones by LRU | +| Subagent sessions | 8 | Only completed ones are evicted; running subagents never — with no room, spawning is rejected | + +## Approval + +Every complete `tool_call` triggers exactly one approval decision: + +```ts +type ApproveFn = (toolCall: OmniMessage) => Promise<"allow" | "deny">; +``` + +| Surface | Behavior | +| --- | --- | +| SDK | Pass `approve` per `session.run`; with none injected the engine denies by default (conservative — nothing gets approved unattended) | +| CLI | `--approve` takes four modes: allow-all (default) / deny-all / read-only / always-ask; read-only auto-approves `permission: "r"` tools and defers the rest to a human | +| Web / Server | The same four modes, set per Session; the mode is re-read from the DB on every decision, so changes take effect immediately; manual decisions arrive via the API | + +A deny produces a synthetic aborted `tool_call_output` (`Tool call denied by user.`) for the model to react to. Every decision is written to the Trace as an `approval_decision` event, forming a complete audit record. Approval happens in the tool-execution phase of the [Agent Loop](/agent-loop). + +## Custom tools & MCP + +The `tools.builtin` array in `system_config.yaml` declares the toolset with entries of the same `ToolDefinitionConfig` shape. The semantics are **wholesale replacement, not merging**: omit the section entirely to keep the full default toolset; once written, the default list is replaced and every tool you keep must carry its complete definition (including the `parameters` JSON Schema — a tool's schema comes entirely from config). `tools.mcpServers` carries MCP server configs (name + config) — enumerating concrete MCP tools is reserved for a later adapter layer and not yet wired. See [Configuration](/configuration). + +```yaml +tools: + # Writing builtin replaces the default toolset wholesale (this example deliberately + # keeps a minimal single-tool set). + builtin: + - name: exec_command + description: Run a shell command in the workspace. + permission: rw + timeoutMs: 120000 + maxOutputLength: 16000 + # parameters: the complete JSON Schema is required (see the default definition + # in packages/core/src/state/default-config.ts); elided here. + mcpServers: [] +``` diff --git a/packages/docs/content/tools.zh.md b/packages/docs/content/tools.zh.md new file mode 100644 index 0000000..06b7570 --- /dev/null +++ b/packages/docs/content/tools.zh.md @@ -0,0 +1,187 @@ +--- +title: 工具与审批 +description: 极简内置工具集的设计与执行契约、Environment 统一收尾规则,以及逐调用审批与 Trace 审计。 +--- + +## 设计取向 + +PenguinHarness 刻意维持一个极小的内置工具集:Shell 是通用接口,文件的读取、写入、编辑全部经由 `exec_command` 完成,不设专门的文件工具。工具越少,注入的 schema 越少,Token 开销越小,模型误调用的概率也越低。 + +## 执行契约 + +所有内置工具实现同一个 `BuiltinTool` 接口(`packages/core/src/environment/tools/types.ts`): + +```ts +interface BuiltinTool { + name: string; + definition: ToolDefinitionConfig; + execute( + args: Record, + ctx: ToolExecutionContext, + ): AsyncGenerator; +} + +interface ToolExecutionContext { + workspaceDir: string; + toolCallId: string; + signal?: AbortSignal; + approve?: ApproveFn; // 供需要派生子 Session 的工具转发(审批继承) +} + +interface ToolResult { + stopReason?: StopReason; // 工具自报终态(优先级最低,见下) + note?: string; // 追加在截断范围之外的终止标记(如退出码) + images?: string[]; // data URL 图像,附加在文本输出之后 +} +``` + +工具本身只需 yield 增量的 `partial_tool_call_output`,收尾由 Environment 集中处理: + +- 流式分帧(start / stop)与 `tool_call_id` 贯穿; +- 超时归并;输出超过 `maxOutputLength`(默认 16000 字符)时截断,保留开头; +- stop_reason 按优先级归并:用户中断 > 超时 > 工具抛错 > 工具自报; +- 输出永不为空:没有任何输出时补 `[no output]`; +- `note`(如退出码)与图像附加在截断范围之外,长输出被截断时终止标记不会丢失。 + +工具与 Environment 从不向引擎抛异常:错误一律折叠为 `tool_call_output` 消息,交给模型阅读并调整下一步。消息结构见 [OmniMessage 协议](/omni-message)。 + +## 配置字段 + +每个工具由一条 `ToolDefinitionConfig` 描述: + +| 字段 | 说明 | +| --- | --- | +| `name` | 工具名,对应模型产出的 `tool_call.name` | +| `description` | 提供给模型的工具说明 | +| `parameters` | 参数 JSON Schema | +| `permission` | `"r"` 只读 / `"rw"` 读写 | +| `forModel` | `"vision"` / `"text-only"`:按 Session 模型类别装配;缺省对所有模型可用 | +| `timeoutMs` | 单次调用超时(ms),默认 120000;`<=0` 关闭 | +| `maxOutputLength` | 输出长度上限(字符);`<=0` 关闭 | + +## 内置工具 + +共 6 个内置工具(装配入口 `packages/core/src/environment/tools/registry.ts`): + +| 工具 | 权限 | 超时(ms) | 用途 | +| --- | --- | --- | --- | +| `exec_command` | rw | 120000 | 在 Workspace 内以 `bash -lc` 运行命令,流式返回 stdout/stderr | +| `input_command` | rw | 130000 | 按 `process_id` 驱动运行中的命令:写 stdin、发 Ctrl-C、轮询输出 | +| `run_subagent` | rw | 600000 | 把自包含子任务委派给同 Workspace 的子 Agent | +| `input_subagent` | rw | 600000 | 轮询后台 Subagent,或在其空闲时追加后续 Prompt | +| `read_image` | r | 60000 | 读取图片并作为图像内容返回(vision 模型) | +| `describe_image` | r | 90000 | 由 `vision_model` 代读图片并返回文字回答(text-only 模型) | + +### 命令会话 + +`exec_command` 先在前台等待;命令超过 `yield_time_ms` 仍未结束时转入后台,返回已有输出和一个 `process_id`,之后用 `input_command` 驱动: + +```text +exec_command(cmd) + ├─ 前台窗口(yield_time_ms,默认 60000)内结束 ──► 完整输出 + 退出码 + └─ 未结束 ──► 转入后台,返回已有输出 + process_id + │ + input_command(process_id[, chars]) ──► 写 stdin / 发 Ctrl-C / 轮询 + └─ 循环驱动,直至命令退出 +``` + +两个工具的参数(明确键名): + +```ts +// exec_command +{ + cmd: string; // 必填:要执行的 shell 命令 + workdir?: string; // 工作目录;缺省为 Workspace 根,相对路径按其解析 + yield_time_ms?: number; // 前台等待时长;默认 60000,最小 250,上限受工具超时约束 +} + +// input_command +{ + process_id: string; // 必填:exec_command 返回的命令会话 id + chars?: string; // 写入 stdin 的字符;单独发送 "\u0003" 传递 Ctrl-C;缺省仅轮询 + yield_time_ms?: number; // 等待时长;有写入默认 250,空轮询默认 5000 +} +``` + +### Subagent + +`run_subagent` 把一段能一次说清的子任务交给子 Agent 执行,同样是两段式:前台窗口(默认 300000ms)过后转入后台并返回 `subagent_id`,由 `input_subagent` 轮询或追加 Prompt;子 Agent 的待审批项会在轮询等待期间浮出。 + +```ts +// run_subagent +{ + prompt: string; // 必填:完整的子任务(含全部上下文与期望的最终产出) + agent_id?: string; // 子 Agent;缺省复用当前 Agent + model_id?: string; // 子 Session 模型;缺省用 Project 默认模型 + yield_time_ms?: number; // 前台等待时长;默认 300000 +} + +// input_subagent +{ + subagent_id: string; // 必填:run_subagent 返回的后台 Subagent id + prompt?: string; // 追加任务,仅在子 Session 空闲时接受;缺省仅轮询 + yield_time_ms?: number; // 等待时长;有追加默认 300000,空轮询默认 10000 +} +``` + +- 深度上限为 1:Subagent 不能再派生 Subagent。 +- 子 Session 继承父 Agent 的审批回调,审批模式随父生效。 +- 子 Session 拥有独立 Trace,父 Trace 以 `subagent` 指针事件链接;子消息带 `origin` 标记回流到父级消息流。见 [Session 与 Trace](/sessions-and-traces)。 + +### 图像工具 + +`read_image` 与 `describe_image` 互斥,按 Session 模型的 vision 标记二选一装配。两者都接受 http(s) URL 或 Workspace 路径,支持 png/jpeg/gif/webp,不超过 5MB。text-only 模型走 `describe_image`:图片连同提问转交 Project 配置的 `vision_model`,其文字回答即工具输出。见 [模型与 Provider](/models)。 + +```ts +// read_image(vision 模型) +{ + source: string; // 必填:http(s) URL,或 Workspace 内的文件路径 +} + +// describe_image(text-only 模型) +{ + source: string; // 必填:同上 + prompt?: string; // 要对图片提出的问题;缺省为详细描述 +} +``` + +### 后台会话上限 + +| 会话类型 | 上限 | 淘汰策略 | +| --- | --- | --- | +| 命令会话 | 64 | 满时优先淘汰已退出者,否则对空闲会话按 LRU 淘汰 | +| Subagent 会话 | 8 | 只淘汰已完成者;运行中的从不淘汰,无空位则拒绝派生 | + +## 审批 + +每个完整的 `tool_call` 触发且只触发一次审批决策: + +```ts +type ApproveFn = (toolCall: OmniMessage) => Promise<"allow" | "deny">; +``` + +| 使用面 | 行为 | +| --- | --- | +| SDK | 每次 `session.run` 传入 `approve` 回调;未注入时引擎默认全部拒绝(保守策略,避免无人值守下误放行) | +| CLI | `--approve` 四种模式:allow-all(默认)/ deny-all / read-only / always-ask;read-only 自动放行 `permission: "r"` 的工具,其余转人工 | +| Web / Server | 同样四种模式,按 Session 设置;每次决策前从数据库重读,改模式立即生效;人工决策经 API 送达 | + +deny 会合成一条 aborted 的 `tool_call_output`(内容为 `Tool call denied by user.`),模型据此调整策略。每次决策都以 `approval_decision` 事件写入 Trace,构成完整的审计记录。审批发生在 [Agent 运行循环](/agent-loop) 的工具执行阶段。 + +## 自定义与 MCP + +`system_config.yaml` 的 `tools.builtin` 数组以 `ToolDefinitionConfig` 同构条目声明工具集。注意语义是**整体替换而非合并**:整段省略时使用完整默认工具集;一旦写出,默认列表即被替换,要保留的每个工具都必须携带完整定义(含 `parameters` JSON Schema——工具的参数 schema 完全来自配置)。`tools.mcpServers` 承载 MCP Server 配置(name + config)——具体 MCP 工具的枚举由后续适配层接管,当前仅保留配置位。见 [配置参考](/configuration)。 + +```yaml +tools: + # 写出 builtin 即整体替换默认工具集(此例刻意只保留一个最小工具集)。 + builtin: + - name: exec_command + description: Run a shell command in the workspace. + permission: rw + timeoutMs: 120000 + maxOutputLength: 16000 + # parameters: 必须携带完整 JSON Schema(默认定义见 + # packages/core/src/state/default-config.ts),此处从略。 + mcpServers: [] +``` diff --git a/packages/docs/content/web-app.en.md b/packages/docs/content/web-app.en.md new file mode 100644 index 0000000..d402199 --- /dev/null +++ b/packages/docs/content/web-app.en.md @@ -0,0 +1,106 @@ +--- +title: Web App Guide +description: A page-by-page guide to the Web App — login, chat, Agent management, models, usage, and traces. +--- + +PenguinHarness ships with a ready-to-use Web App: multi-user login, streaming chat, Agent configuration, model and usage management all happen in the browser. This guide walks through the app page by page. For installation and first launch, see the [Quickstart](/quickstart). + +## Source layout + +```text +packages/web/src +├── api/ # fetch wrapper · one function per API (DTOs type-only from @prismshadow/penguin-server/api) · SSE wrapper +├── state/ # auth / project / sessions / theme / locale contexts +├── lib/omni/ # OmniMessage stream → view-model reducer; connect-first + dedup stream controller +├── components/ # ui primitives (modal / drawer / select …) and the app layout +└── features/ # chat / agents / skills / models / usage / traces / benchmark / admin pages +``` + +## Startup and Login + +```bash +penguin web +# open http://127.0.0.1:7364 +``` + +The initial account is `admin` / `admin123`. There is no self-registration: accounts are created by an admin on the user-management page, and every new user automatically gets an independent initial Project named `-default_project`. While the initial password is still in use, a banner prompts the user to change it. + +Logins persist for 7 days with sliding renewal; an admin password reset invalidates all of that user's login sessions. + +The interface language (中文 / English / system) and theme (light / dark / system) can be switched at any time. + +## Chat (/chat) + +### Creating a Conversation + +A new conversation starts as a draft: pick the Agent, the Workspace (via a server-side directory browser), the approval mode, and the model before sending the first message. The Session is created on first send, and from then on its model and Workspace are locked. + +There are four approval modes: `allow-all`, `deny-all`, `read-only` (only read-only tools pass), and `always-ask`. See [Tools and Approvals](/tools). + +### Streaming Rendering + +- Model text renders token by token; thinking blocks are collapsible; +- Tool cards expand to show arguments and output, with a live timer while running; +- Subagents appear as nested cards; context compaction shows a banner; +- After each Task, a stats line shows tokens, TPS, elapsed time, and cost. + +### Input and Shortcuts + +- Enter sends, Shift+Enter inserts a newline, and images can be pasted; +- Typing `/` opens the slash menu: trigger context compaction (`/compact`) or toggle installed Skills — chosen Skills are sent along with the message in a `` block; +- Typing `@` mentions another Agent to hand the conversation over to it; +- When human approval is required, tool calls show inline allow/deny buttons in the message stream; the approval mode can be changed mid-Session. + +### Files Panel + +The files panel browses the Workspace tree, previews files (Markdown / HTML rendered), uploads files (≤ 14MB each), and downloads them. + +## Agent Management (/agents) + +The list page creates and deletes Agents; clicking through opens the `/agents/:agentId` settings page, organized into tabs: + +| Tab | Contents | +| --- | --- | +| Overview | Basic info, plus export / import of Agent State snapshots | +| Prompt | AGENTS.md and system_prompt | +| Runtime | Runtime parameters such as max_turns, model.*, compaction.* | +| Tools | Built-in tool table and MCP server JSON configuration | +| Vault | Environment-variable entries with masked values | +| Schedule | Scheduled tasks (TOML-defined): create, edit, toggle, delete | + +Scheduled tasks fire on a fixed period (minimum 5 minutes) and run only while the service is running. + +## Skill Library (/skills) + +Browse the Skill library by group, install Skills onto an Agent, or quick-invoke one into a chat draft. + +## Model Configuration (/models) + +A per-Project model table grouped by provider. Models can be added and edited: identity is the `(provider, model_id)` pair, credentials are masked, and context window, pricing, and the vision flag are configurable. You can set the default model and the vision model (which reads images on behalf of session models without image input), and run a connectivity test on any entry. Only Project owners can edit. For concepts, see [Models and Providers](/models). + +## Usage (/usage) + +- Filters: Agent, model, date range; +- Summary cards: today / last 7 days / cumulative; +- Charts: per-Agent share, per-model success rates, daily Token and cost trends; +- A server error panel summarizing recent server-side error records. + +## Trace Browser (/traces) + +Drill down Agent → date → Session → Trace file. Per-turn cards show a context-occupancy donut and a cache breakdown, alongside a lane-based execution timeline and the full event list. For the storage model, see [Sessions and Traces](/sessions-and-traces). + +## Benchmark (/benchmark) + +Read-only scoreboards per Benchmark: switch the metric (score / cost / duration), drill into each Case's runs, and jump to the linked Session and Trace. Works together with the [Self-Improvement](/self-improvement) workflow. + +## User Administration (/admin/users) + +Admin only: list and create users, reset passwords, and delete users (the built-in admin cannot be deleted). + +## Projects and Members + +The sidebar provides a Project switcher and supports creating new Projects. Members have two roles, owner and member: owners manage membership and exclusively edit models, Vault, and Schedules, as well as perform deletions. + +## Production Deployment + +The server hosts the built SPA itself (same origin, SPA fallback), so a single `penguin web` or `penguin server` process is all production needs. The npm package bundles the frontend build; to serve a custom static directory, override it with `PENGUIN_WEB_DIST` — see the [Configuration Reference](/configuration). diff --git a/packages/docs/content/web-app.zh.md b/packages/docs/content/web-app.zh.md new file mode 100644 index 0000000..b909439 --- /dev/null +++ b/packages/docs/content/web-app.zh.md @@ -0,0 +1,106 @@ +--- +title: Web App 指南 +description: 按页面组织的 Web App 使用指南:登录、Chat、Agent 管理、模型、用量与 Trace。 +--- + +PenguinHarness 自带一个开箱即用的 Web App:多用户登录、流式对话、Agent 配置、模型与用量管理都在浏览器中完成。本文按页面组织,逐一介绍各页面的功能与操作。安装与首次启动见[快速开始](/quickstart)。 + +## 目录结构 + +```text +packages/web/src +├── api/ # fetch 封装 · 每个 API 一个函数(DTO type-only 来自 @prismshadow/penguin-server/api)· SSE 封装 +├── state/ # auth / project / sessions / theme / locale 五个 context +├── lib/omni/ # OmniMessage 流 → 渲染视图模型 reducer;连接先行 + 去重的流控制器 +├── components/ # ui 原语(modal / drawer / select …)与应用布局 +└── features/ # chat / agents / skills / models / usage / traces / benchmark / admin 各页面 +``` + +## 启动与登录 + +```bash +penguin web +# 打开 http://127.0.0.1:7364 +``` + +初始账号为 `admin` / `admin123`。系统不开放自助注册:账号由管理员在用户管理页创建;每个新用户会自动获得一个独立的初始 Project,命名为 `-default_project`。仍在使用初始密码时,页面会以横幅提示尽快修改。 + +登录状态保持 7 天(滑动续期);管理员重置密码会使该用户的全部登录会话失效。 + +界面语言(中文 / English / 跟随系统)与主题(浅色 / 深色 / 跟随系统)可随时切换。 + +## Chat 页面(/chat) + +### 新建会话 + +新会话从草稿开始:先选择 Agent、Workspace(服务器端目录浏览器选取)、审批模式与模型,再发送第一条消息。Session 在首次发送时才真正创建,此后该会话的模型与 Workspace 即被锁定。 + +审批模式共四种:`allow-all`(全部放行)、`deny-all`(全部拒绝)、`read-only`(仅放行只读工具)、`always-ask`(每次询问),详见[工具与审批](/tools)。 + +### 流式渲染 + +- 模型文本逐 Token 渲染,思考块可折叠; +- 工具卡片可展开查看参数与输出,执行中显示实时计时; +- 子 Agent 以嵌套卡片呈现;上下文压缩以横幅提示; +- 每个 Task 结束后显示统计行:Token 用量、TPS、耗时与费用。 + +### 输入与快捷操作 + +- Enter 发送,Shift+Enter 换行,支持粘贴图片; +- 输入 `/` 打开快捷菜单:触发上下文压缩(`/compact`),或勾选已安装的 Skill——所选 Skill 会以 `` 块随消息发送; +- 输入 `@` 提及其他 Agent,将会话交接给它; +- 需要人工审批时,工具调用在消息流中内联显示“允许 / 拒绝”按钮;审批模式在会话中途可随时调整。 + +### 文件面板 + +文件面板可浏览 Workspace 目录树、预览文件(Markdown / HTML 渲染显示)、上传文件(单个 ≤ 14MB)与下载文件。 + +## Agent 管理(/agents) + +列表页支持创建与删除 Agent;点击进入 `/agents/:agentId` 设置页,按标签页组织: + +| 标签页 | 内容 | +| --- | --- | +| Overview | 基本信息,以及 Agent State 快照的导出 / 导入 | +| Prompt | AGENTS.md 与 system_prompt | +| Runtime | max_turns、model.*、compaction.* 等运行参数 | +| Tools | 内置工具表格与 MCP Server 的 JSON 配置 | +| Vault | 环境变量条目,值以掩码显示 | +| Schedule | 定时任务(TOML 定义):创建、编辑、启停、删除 | + +定时任务按固定周期触发(最短 5 分钟),且仅在服务运行期间执行。 + +## Skill 库(/skills) + +按分组浏览 Skill 库,可将 Skill 安装到指定 Agent,或一键带入 Chat 草稿快速调用。 + +## 模型配置(/models) + +按 Provider 分组展示当前 Project 的模型表格。支持添加与编辑模型:以 `(provider, model_id)` 为唯一标识,凭据以掩码显示,可配置上下文窗口、定价与视觉(vision)标记;可设置默认模型与视觉模型(在会话模型不支持图片输入时代为读图),并对任一模型做连通性测试。仅 Project Owner 可编辑,概念说明见[模型与 Provider](/models)。 + +## 用量统计(/usage) + +- 筛选条件:Agent、模型、日期范围; +- 概览卡片:今日 / 近 7 天 / 累计用量; +- 图表:各 Agent 占比、各模型成功率、每日 Token 与费用趋势; +- 服务端错误面板:汇总最近的服务端错误记录。 + +## Trace 浏览(/traces) + +按 Agent → 日期 → Session → Trace 文件逐级下钻。每回合卡片展示上下文占用环形图与缓存构成,并提供泳道式执行时间线与完整事件列表。Trace 的存储模型见 [Session 与 Trace](/sessions-and-traces)。 + +## Benchmark(/benchmark) + +只读展示各 Benchmark 的评分板,可切换指标(得分 / 费用 / 耗时),下钻查看每个 Case 的多次运行结果,并跳转到关联的 Session 与 Trace。配合[自我进化](/self-improvement)工作流使用。 + +## 用户管理(/admin/users) + +仅管理员可见:列出与创建用户、重置密码、删除用户(内置 admin 不可删除)。 + +## Project 与成员 + +侧边栏提供 Project 切换器,并支持创建新 Project。成员分为 Owner 与 Member 两种角色:Owner 负责成员管理,并独占模型、Vault、Schedule 的编辑以及各类删除操作。 + +## 生产部署 + +服务端自身托管构建好的 SPA(同源、SPA fallback),生产环境只需运行 `penguin web` 或 `penguin server` 一个进程。npm 安装包已内置前端产物;如需自定义静态目录,可用 `PENGUIN_WEB_DIST` 覆盖,见[配置参考](/configuration)。 diff --git a/packages/docs/index.html b/packages/docs/index.html new file mode 100644 index 0000000..98be81d --- /dev/null +++ b/packages/docs/index.html @@ -0,0 +1,35 @@ + + + + + + + + + + + + + + PenguinHarness Docs + + + + +

+ + + diff --git a/packages/docs/package.json b/packages/docs/package.json new file mode 100644 index 0000000..518487e --- /dev/null +++ b/packages/docs/package.json @@ -0,0 +1,32 @@ +{ + "name": "@prismshadow/penguin-docs", + "version": "0.0.1", + "private": true, + "type": "module", + "description": "PenguinHarness documentation site (React + Vite + Tailwind CSS): bilingual zh/en Markdown pages with light/dark themes and per-page Copy Markdown, deployed to GitHub Pages under /docs/ next to the landing page.", + "scripts": { + "dev": "vite", + "build": "vite build && node scripts/postbuild.mjs", + "preview": "vite preview", + "typecheck": "tsc --noEmit -p tsconfig.json", + "test": "vitest run --passWithNoTests" + }, + "dependencies": { + "react": "^19.1.0", + "react-dom": "^19.1.0", + "react-markdown": "^10.1.0", + "react-router": "^7.6.0", + "remark-gfm": "^4.0.0" + }, + "devDependencies": { + "@tailwindcss/vite": "^4.1.0", + "@types/node": "^24.0.0", + "@types/react": "^19.1.0", + "@types/react-dom": "^19.1.0", + "@vitejs/plugin-react": "^4.5.0", + "tailwindcss": "^4.1.0", + "typescript": "^5.6.0", + "vite": "^7.0.0", + "vitest": "^2.1.0" + } +} diff --git a/packages/docs/public/penguin-logo.svg b/packages/docs/public/penguin-logo.svg new file mode 100644 index 0000000..0d48619 --- /dev/null +++ b/packages/docs/public/penguin-logo.svg @@ -0,0 +1 @@ + diff --git a/packages/docs/scripts/postbuild.mjs b/packages/docs/scripts/postbuild.mjs new file mode 100644 index 0000000..d0524d3 --- /dev/null +++ b/packages/docs/scripts/postbuild.mjs @@ -0,0 +1,25 @@ +/** + * Post-build step: GitHub Pages serves static files only, so every doc route gets a + * copy of the SPA shell at dist//index.html — deep links (…/docs/omni-message) + * then load without relying on a 404 fallback (the site-root 404.html belongs to the + * landing page, which would swallow /docs/* misses). Slugs are derived from the + * content/ filenames (..md), the same source the router reads. + */ +import { copyFileSync, mkdirSync, readdirSync } from "node:fs"; +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; + +const pkg = join(dirname(fileURLToPath(import.meta.url)), ".."); +const dist = join(pkg, "dist"); + +const slugs = new Set( + readdirSync(join(pkg, "content")) + .map((file) => /^(.+)\.(zh|en)\.md$/.exec(file)?.[1]) + .filter((slug) => slug !== undefined), +); + +for (const slug of slugs) { + mkdirSync(join(dist, slug), { recursive: true }); + copyFileSync(join(dist, "index.html"), join(dist, slug, "index.html")); +} +console.log(`[postbuild] wrote ${slugs.size} route shells under dist/`); diff --git a/packages/docs/src/app.tsx b/packages/docs/src/app.tsx new file mode 100644 index 0000000..f4c3a2a --- /dev/null +++ b/packages/docs/src/app.tsx @@ -0,0 +1,20 @@ +/** + * App root: Locale -> Theme -> LocaleScope -> Router provider composition, same as + * the landing page (LocaleScope remounts the tree keyed by locale so every `S.x` + * read reflects the active language). + */ +import { LocaleProvider, LocaleScope } from "./state/locale"; +import { ThemeProvider } from "./state/theme"; +import { AppRouter } from "./router"; + +export function App() { + return ( + + + + + + + + ); +} diff --git a/packages/docs/src/components/copy-markdown-button.tsx b/packages/docs/src/components/copy-markdown-button.tsx new file mode 100644 index 0000000..02da289 --- /dev/null +++ b/packages/docs/src/components/copy-markdown-button.tsx @@ -0,0 +1,54 @@ +/** + * "Copy Markdown" button: puts the page's Markdown source on the clipboard with a + * transient "copied" state — so a page can be pasted into a model context, an issue + * or a note as clean Markdown rather than rendered HTML. + */ +import { useEffect, useRef, useState } from "react"; +import { S } from "../lib/strings"; +import { CheckIcon, CopyIcon } from "./icons"; + +export function CopyMarkdownButton({ text }: { text: string }) { + const [copied, setCopied] = useState(false); + const timer = useRef | null>(null); + + useEffect( + () => () => { + if (timer.current) clearTimeout(timer.current); + }, + [], + ); + + const onCopy = async () => { + try { + await navigator.clipboard.writeText(text); + } catch { + // Clipboard API unavailable (e.g. non-secure context): fall back to a hidden textarea. + const ta = document.createElement("textarea"); + ta.value = text; + document.body.appendChild(ta); + ta.select(); + document.execCommand("copy"); + ta.remove(); + } + setCopied(true); + if (timer.current) clearTimeout(timer.current); + timer.current = setTimeout(() => setCopied(false), 1600); + }; + + return ( + + ); +} diff --git a/packages/docs/src/components/footer.tsx b/packages/docs/src/components/footer.tsx new file mode 100644 index 0000000..2a7c804 --- /dev/null +++ b/packages/docs/src/components/footer.tsx @@ -0,0 +1,25 @@ +/** Slim docs footer: copyright + repo / license / main-site links on one line. */ +import { S } from "../lib/strings"; +import { LICENSE_URL, REPO_URL, SITE_URL } from "../lib/links"; + +export function Footer() { + const link = "transition-colors hover:text-gray-900 dark:hover:text-gray-100 whitespace-nowrap"; + return ( + + ); +} diff --git a/packages/docs/src/components/icons.tsx b/packages/docs/src/components/icons.tsx new file mode 100644 index 0000000..21c1f71 --- /dev/null +++ b/packages/docs/src/components/icons.tsx @@ -0,0 +1,125 @@ +/** + * Inline icon set (lucide-style 24x24 stroke icons + the GitHub mark), the subset of + * the landing page's set that the docs UI needs. Kept local so the docs site has zero + * icon dependencies; all icons inherit currentColor. + */ +import type { ReactNode, SVGProps } from "react"; + +type IconProps = SVGProps; + +function Icon({ children, ...props }: IconProps & { children: ReactNode }) { + return ( + + ); +} + +export function GitHubIcon(props: IconProps) { + return ( + + ); +} + +export function SunIcon(props: IconProps) { + return ( + + + + + ); +} + +export function MoonIcon(props: IconProps) { + return ( + + + + ); +} + +export function MonitorIcon(props: IconProps) { + return ( + + + + + ); +} + +export function CopyIcon(props: IconProps) { + return ( + + + + + ); +} + +export function CheckIcon(props: IconProps) { + return ( + + + + ); +} + +export function MenuIcon(props: IconProps) { + return ( + + + + ); +} + +export function XIcon(props: IconProps) { + return ( + + + + ); +} + +export function ArrowRightIcon(props: IconProps) { + return ( + + + + ); +} + +export function GlobeIcon(props: IconProps) { + return ( + + + + + ); +} + +export function ChevronDownIcon(props: IconProps) { + return ( + + + + ); +} + +export function ExternalLinkIcon(props: IconProps) { + return ( + + + + ); +} diff --git a/packages/docs/src/components/lang-toggle.tsx b/packages/docs/src/components/lang-toggle.tsx new file mode 100644 index 0000000..dfe92a7 --- /dev/null +++ b/packages/docs/src/components/lang-toggle.tsx @@ -0,0 +1,74 @@ +/** + * Language menu: 中文 / English / follow system, persisted via the locale context. + * A small dropdown (globe + current label); closes on outside click or selection. + * Scroll position across the locale remount is preserved by LocaleScope. + */ +import { useEffect, useRef, useState } from "react"; +import { useLocale } from "../state/locale"; +import type { LangPref } from "../state/locale"; +import { S } from "../lib/strings"; +import { CheckIcon, ChevronDownIcon, GlobeIcon } from "./icons"; + +export function LangToggle() { + const { lang, setLang } = useLocale(); + const [open, setOpen] = useState(false); + const ref = useRef(null); + + useEffect(() => { + if (!open) return; + const onDown = (e: MouseEvent) => { + if (ref.current && !ref.current.contains(e.target as Node)) setOpen(false); + }; + document.addEventListener("mousedown", onDown); + return () => document.removeEventListener("mousedown", onDown); + }, [open]); + + const OPTIONS: Array<{ value: LangPref; label: string }> = [ + { value: "en", label: S.lang.en }, + { value: "zh", label: S.lang.zh }, + { value: "system", label: S.lang.system }, + ]; + const current = OPTIONS.find((o) => o.value === lang) ?? OPTIONS[2]!; + + return ( +
+ + {open && ( +
+ {OPTIONS.map((o) => ( + + ))} +
+ )} +
+ ); +} diff --git a/packages/docs/src/components/nav.tsx b/packages/docs/src/components/nav.tsx new file mode 100644 index 0000000..616ba0f --- /dev/null +++ b/packages/docs/src/components/nav.tsx @@ -0,0 +1,59 @@ +/** + * Sticky top bar: logo + site name + "Docs" badge, then (right) a link back to the + * main site, GitHub, language/theme toggles and — on small screens — the sidebar + * toggle. The sidebar itself lives in the layout (router.tsx); this bar only flips + * its open state. + */ +import { Link } from "react-router"; +import { S } from "../lib/strings"; +import { REPO_URL, SITE_URL } from "../lib/links"; +import { GitHubIcon, MenuIcon, XIcon } from "./icons"; +import { ThemeToggle } from "./theme-toggle"; +import { LangToggle } from "./lang-toggle"; + +export function Nav({ menuOpen, onToggleMenu }: { menuOpen: boolean; onToggleMenu: () => void }) { + return ( +
+
+ + + + + {S.siteName} + + {S.docsBadge} + + + + +
+
+ ); +} diff --git a/packages/docs/src/components/sidebar.tsx b/packages/docs/src/components/sidebar.tsx new file mode 100644 index 0000000..411fc05 --- /dev/null +++ b/packages/docs/src/components/sidebar.tsx @@ -0,0 +1,49 @@ +/** + * Docs sidebar: sections + page links from DOCS_NAV, titles resolved from the active + * locale's frontmatter. Desktop: sticky column. Mobile: the layout renders it as an + * overlay panel under the top bar; onNavigate closes that panel. + */ +import { Link, useLocation } from "react-router"; +import { S } from "../lib/strings"; +import { useLocale } from "../state/locale"; +import { DOCS_NAV, HOME_SLUG } from "../lib/nav"; +import { docTitle } from "../lib/docs"; + +export function Sidebar({ onNavigate }: { onNavigate?: () => void }) { + const { locale } = useLocale(); + const { pathname } = useLocation(); + const activeSlug = pathname.replace(/^\/|\/$/g, "") || HOME_SLUG; + + return ( + + ); +} diff --git a/packages/docs/src/components/theme-toggle.tsx b/packages/docs/src/components/theme-toggle.tsx new file mode 100644 index 0000000..4d552ee --- /dev/null +++ b/packages/docs/src/components/theme-toggle.tsx @@ -0,0 +1,24 @@ +/** Theme cycle button: light -> dark -> system, icon reflects the current mode. */ +import { useTheme } from "../state/theme"; +import type { ThemeMode } from "../state/theme"; +import { S } from "../lib/strings"; +import { MonitorIcon, MoonIcon, SunIcon } from "./icons"; + +const NEXT: Record = { light: "dark", dark: "system", system: "light" }; + +export function ThemeToggle() { + const { mode, setMode } = useTheme(); + const label = mode === "light" ? S.theme.light : mode === "dark" ? S.theme.dark : S.theme.system; + const IconCmp = mode === "light" ? SunIcon : mode === "dark" ? MoonIcon : MonitorIcon; + return ( + + ); +} diff --git a/packages/docs/src/lib/docs.ts b/packages/docs/src/lib/docs.ts new file mode 100644 index 0000000..d8f344d --- /dev/null +++ b/packages/docs/src/lib/docs.ts @@ -0,0 +1,60 @@ +/** + * Docs index: local Markdown pages imported at build time via import.meta.glob. + * File naming: content/..md — one file per page per language; a page + * missing the active language falls back to the other one, so navigation is always + * complete in both locales. Same architecture as the landing page blog. + */ +import { parseFrontmatter } from "./frontmatter"; +import type { Locale } from "../state/locale"; + +export interface DocPage { + slug: string; + lang: Locale; + title: string; + /** One-line summary rendered under the title (optional). */ + description: string; + body: string; +} + +const files = import.meta.glob("../../content/*.md", { + query: "?raw", + import: "default", + eager: true, +}) as Record; + +function toDoc(path: string, raw: string): DocPage | null { + const file = path.split("/").pop() ?? ""; + const match = /^(.+)\.(zh|en)\.md$/.exec(file); + if (!match) return null; + const { meta, body } = parseFrontmatter(raw); + return { + slug: match[1]!, + lang: match[2] as Locale, + title: meta.title ?? match[1]!, + description: meta.description ?? "", + body, + }; +} + +const ALL: DocPage[] = Object.entries(files) + .map(([path, raw]) => toDoc(path, raw)) + .filter((doc): doc is DocPage => doc !== null); + +/** The locale's version of a page (fallback to the other language). */ +export function getDoc(slug: string, locale: Locale): DocPage | undefined { + const candidates = ALL.filter((doc) => doc.slug === slug); + return candidates.find((doc) => doc.lang === locale) ?? candidates[0]; +} + +/** Localized page title for sidebar / pagination labels. */ +export function docTitle(slug: string, locale: Locale): string { + return getDoc(slug, locale)?.title ?? slug; +} + +/** + * The page as plain Markdown (title heading + body) — what the per-page + * "Copy Markdown" button puts on the clipboard. + */ +export function docMarkdown(doc: DocPage): string { + return `# ${doc.title}\n\n${doc.body}\n`; +} diff --git a/packages/docs/src/lib/frontmatter.ts b/packages/docs/src/lib/frontmatter.ts new file mode 100644 index 0000000..9173559 --- /dev/null +++ b/packages/docs/src/lib/frontmatter.ts @@ -0,0 +1,31 @@ +/** + * Minimal frontmatter parser for doc pages: a leading `---` block of `key: value` + * lines (values may contain colons; quotes optional). Kept dependency-free and pure + * so it is unit-testable without Vite. Same format as the landing page blog. + */ + +export interface Frontmatter { + meta: Record; + body: string; +} + +export function parseFrontmatter(raw: string): Frontmatter { + const normalized = raw.replace(/\r\n/g, "\n"); + const match = /^---\n([\s\S]*?)\n---\n?/.exec(normalized); + if (!match) return { meta: {}, body: normalized.trim() }; + const meta: Record = {}; + for (const line of match[1]!.split("\n")) { + const idx = line.indexOf(":"); + if (idx === -1) continue; + const key = line.slice(0, idx).trim(); + let value = line.slice(idx + 1).trim(); + if ( + (value.startsWith('"') && value.endsWith('"')) || + (value.startsWith("'") && value.endsWith("'")) + ) { + value = value.slice(1, -1); + } + if (key) meta[key] = value; + } + return { meta, body: normalized.slice(match[0].length).trim() }; +} diff --git a/packages/docs/src/lib/links.ts b/packages/docs/src/lib/links.ts new file mode 100644 index 0000000..5d10fae --- /dev/null +++ b/packages/docs/src/lib/links.ts @@ -0,0 +1,12 @@ +/** External links and language-independent constants used across the docs site. */ + +export const REPO_URL = "https://github.com/Prism-Shadow/penguin-harness"; +export const LICENSE_URL = `${REPO_URL}/blob/main/LICENSE`; + +/** + * The main site sits one level above the docs (both ship in one GitHub Pages + * artifact: landing at "//", docs at "//docs/"). In local dev the docs + * base is "/" so this resolves to the docs root itself — the landing page runs on + * its own dev server there. + */ +export const SITE_URL = import.meta.env.BASE_URL.replace(/docs\/$/, ""); diff --git a/packages/docs/src/lib/nav.ts b/packages/docs/src/lib/nav.ts new file mode 100644 index 0000000..3dd6cac --- /dev/null +++ b/packages/docs/src/lib/nav.ts @@ -0,0 +1,43 @@ +/** + * Docs navigation: the single source of truth for sidebar sections, page order and + * prev/next pagination. Section labels live in the strings dictionaries (S.sections); + * page titles come from each Markdown file's frontmatter. Kept pure (no import.meta) + * so the content-integrity test can import it under plain node. + */ + +export interface DocsSectionDef { + /** Section id — also the key into S.sections for the localized label. */ + id: "start" | "design" | "guides" | "reference"; + /** Page slugs in display order; content files are content/..md. */ + slugs: string[]; +} + +export const DOCS_NAV: DocsSectionDef[] = [ + { id: "start", slugs: ["introduction", "installation", "quickstart"] }, + { + id: "design", + slugs: [ + "architecture", + "omni-message", + "agent-loop", + "message-flow", + "interfaces", + "tools", + "skills", + "models", + "sessions-and-traces", + ], + }, + { id: "guides", slugs: ["web-app", "self-improvement"] }, + { id: "reference", slugs: ["cli", "server-api", "configuration"] }, +]; + +/** All slugs in display order (pagination order). */ +export const DOC_SLUGS: string[] = DOCS_NAV.flatMap((section) => section.slugs); + +/** The docs landing page ("/" renders this slug). */ +export const HOME_SLUG = DOC_SLUGS[0]!; + +export function sectionOf(slug: string): DocsSectionDef | undefined { + return DOCS_NAV.find((section) => section.slugs.includes(slug)); +} diff --git a/packages/docs/src/lib/strings-en.ts b/packages/docs/src/lib/strings-en.ts new file mode 100644 index 0000000..100bee4 --- /dev/null +++ b/packages/docs/src/lib/strings-en.ts @@ -0,0 +1,52 @@ +/** English dictionary for the docs UI (same shape as `zh` in strings.ts). */ +import type { Strings } from "./strings"; + +export const en: Strings = { + siteName: "PenguinHarness", + docsBadge: "Docs", + + nav: { + home: "Website", + github: "GitHub", + openMenu: "Open navigation", + closeMenu: "Close navigation", + }, + + theme: { + label: "Theme", + light: "Light", + dark: "Dark", + system: "System", + }, + + lang: { + label: "Language", + zh: "中文", + en: "English", + system: "System", + }, + + sections: { + start: "Get Started", + design: "Core Design", + guides: "Guides", + reference: "Reference", + } as Record, + + doc: { + toc: "On this page", + copyMarkdown: "Copy Markdown", + copied: "Copied", + prev: "Previous", + next: "Next", + notFound: "Page not found", + backHome: "Back to docs home", + }, + + footer: { + repo: "GitHub repository", + license: "Apache-2.0 License", + site: "Website", + copyright: "© 2026 Prism Shadow · Open source under Apache-2.0", + }, +}; diff --git a/packages/docs/src/lib/strings.ts b/packages/docs/src/lib/strings.ts new file mode 100644 index 0000000..99ed19b --- /dev/null +++ b/packages/docs/src/lib/strings.ts @@ -0,0 +1,70 @@ +/** + * Docs UI copy (bilingual): this file holds the Chinese dictionary `zh` and the runtime + * active dictionary `S`; the English dictionary lives in strings-en.ts (constrained to + * the same shape by the `Strings` type). Locale switching is handled by state/locale.tsx, + * which calls `setActiveStrings` and remounts the tree keyed by locale — keep `S.x` + * reads inside components. Doc page bodies are Markdown files under content/, not here. + */ +export const zh = { + siteName: "PenguinHarness", + docsBadge: "Docs", + + nav: { + home: "产品主页", + github: "GitHub", + openMenu: "打开目录", + closeMenu: "关闭目录", + }, + + theme: { + label: "主题", + light: "浅色", + dark: "深色", + system: "跟随系统", + }, + + lang: { + label: "语言", + zh: "中文", + en: "English", + system: "跟随系统", + }, + + sections: { + start: "开始", + design: "核心设计", + guides: "使用指南", + reference: "参考", + } as Record, + + doc: { + toc: "本页目录", + copyMarkdown: "复制 Markdown", + copied: "已复制", + prev: "上一页", + next: "下一页", + notFound: "页面不存在", + backHome: "返回文档首页", + }, + + footer: { + repo: "GitHub 仓库", + license: "Apache-2.0 License", + site: "产品主页", + copyright: "© 2026 Prism Shadow · 基于 Apache-2.0 协议开源", + }, +}; + +/** Dictionary shape (constrains the English dictionary so keys line up). */ +export type Strings = typeof zh; + +/** + * Runtime active dictionary (live binding): the locale Provider calls setActiveStrings + * to switch before render, and remounts the whole tree keyed by locale so every `S.x` + * read reflects the current language. + */ +export let S: Strings = zh; + +export function setActiveStrings(next: Strings): void { + S = next; +} diff --git a/packages/docs/src/lib/toc.ts b/packages/docs/src/lib/toc.ts new file mode 100644 index 0000000..b9bb459 --- /dev/null +++ b/packages/docs/src/lib/toc.ts @@ -0,0 +1,42 @@ +/** + * Table-of-contents helpers: extract ##/### headings from a Markdown body (skipping + * fenced code blocks) and slugify them the same way the rendered headings do, so TOC + * anchors and heading ids always match. Pure and unit-testable; same behavior as the + * landing page blog. + */ + +export interface TocEntry { + id: string; + text: string; + depth: 2 | 3; +} + +/** Heading text -> anchor id (keeps CJK, lowercases latin, hyphenates spaces). */ +export function slugifyHeading(text: string): string { + return text + .trim() + .toLowerCase() + .replace(/[^\p{L}\p{N}\s-]/gu, "") + .replace(/\s+/g, "-"); +} + +export function extractToc(body: string): TocEntry[] { + const entries: TocEntry[] = []; + let inFence = false; + for (const line of body.split("\n")) { + if (/^\s*(```|~~~)/.test(line)) { + inFence = !inFence; + continue; + } + if (inFence) continue; + const match = /^(#{2,3})\s+(.+?)\s*$/.exec(line); + if (!match) continue; + const text = match[2]!; + entries.push({ + id: slugifyHeading(text), + text, + depth: match[1]!.length === 2 ? 2 : 3, + }); + } + return entries; +} diff --git a/packages/docs/src/main.tsx b/packages/docs/src/main.tsx new file mode 100644 index 0000000..35f19ca --- /dev/null +++ b/packages/docs/src/main.tsx @@ -0,0 +1,14 @@ +/** Docs entry point: mounts the React root component. */ +import { StrictMode } from "react"; +import { createRoot } from "react-dom/client"; +import { App } from "./app"; +import "./styles.css"; + +const container = document.getElementById("root"); +if (!container) throw new Error("#root mount point not found"); + +createRoot(container).render( + + + , +); diff --git a/packages/docs/src/pages/doc-page.tsx b/packages/docs/src/pages/doc-page.tsx new file mode 100644 index 0000000..16ec612 --- /dev/null +++ b/packages/docs/src/pages/doc-page.tsx @@ -0,0 +1,284 @@ +/** + * Doc page: renders the Markdown body (react-markdown + GFM) in .md-body style with a + * sticky "on this page" TOC on wide screens, a per-page Copy Markdown button, and + * prev/next pagination following the sidebar order. Headings get slug ids (same + * slugifier as the TOC) and the active section is tracked while scrolling — the same + * mechanics as the landing page blog. Internal links written as "/" navigate + * client-side; external links open in a new tab. + */ +import { useCallback, useEffect, useMemo, useRef, useState } from "react"; +import type { ReactNode } from "react"; +import Markdown from "react-markdown"; +import remarkGfm from "remark-gfm"; +import { Link, useParams } from "react-router"; +import { S } from "../lib/strings"; +import { useLocale } from "../state/locale"; +import { docMarkdown, docTitle, getDoc } from "../lib/docs"; +import { DOC_SLUGS, HOME_SLUG, sectionOf } from "../lib/nav"; +import { extractToc, slugifyHeading } from "../lib/toc"; +import { CopyMarkdownButton } from "../components/copy-markdown-button"; +import { ArrowRightIcon } from "../components/icons"; + +/** Flatten react-markdown heading children to plain text for slugging. */ +function nodeText(node: ReactNode): string { + if (node === null || node === undefined || typeof node === "boolean") return ""; + if (typeof node === "string" || typeof node === "number") return String(node); + if (Array.isArray(node)) return node.map(nodeText).join(""); + if (typeof node === "object" && "props" in node) { + return nodeText((node as { props: { children?: ReactNode } }).props.children); + } + return ""; +} + +function Toc({ + entries, + activeId, + onNavigate, +}: { + entries: ReturnType; + activeId: string; + onNavigate: (id: string) => void; +}) { + return ( + + ); +} + +function Pager({ slug }: { slug: string }) { + const { locale } = useLocale(); + const index = DOC_SLUGS.indexOf(slug); + if (index === -1) return null; + const prev = index > 0 ? DOC_SLUGS[index - 1]! : null; + const next = index < DOC_SLUGS.length - 1 ? DOC_SLUGS[index + 1]! : null; + const card = (target: string, dir: "prev" | "next") => ( + + + {dir === "prev" && } + {dir === "prev" ? S.doc.prev : S.doc.next} + {dir === "next" && } + + + {docTitle(target, locale)} + + + ); + return ( +
+ {prev && card(prev, "prev")} + {next && card(next, "next")} +
+ ); +} + +/** Internal "/" links -> client-side navigation; external links -> new tab. */ +function MdLink({ href = "", children }: { href?: string; children?: ReactNode }) { + if (href.startsWith("/")) { + return {children}; + } + if (/^https?:\/\//.test(href)) { + return ( + + {children} + + ); + } + return {children}; +} + +export function DocPage() { + const { slug = HOME_SLUG } = useParams(); + const { locale } = useLocale(); + const doc = getDoc(slug, locale); + const section = sectionOf(slug); + const toc = useMemo(() => (doc ? extractToc(doc.body) : []), [doc]); + const [activeId, setActiveId] = useState(""); + /** + * A TOC click (or an initial #hash) pins the target entry as active: a short tail + * section can never reach the reading line, so pure position tracking would highlight + * a neighbor instead of what the user just chose. The pin releases on the first real + * scroll gesture (wheel / touch / scroll keys) — programmatic smooth scrolling fires + * only `scroll` events, so it never unpins by itself. + */ + const pinnedId = useRef(null); + + const pinTo = useCallback((id: string) => { + pinnedId.current = id; + setActiveId(id); + }, []); + + // Track the heading last crossed by a moving reading line for TOC highlighting. + // The line sits 100px under the sticky header at the top of the page and slides down + // to ~40px above the viewport bottom at full scroll: it is monotonic in scrollY, so + // every heading gets a highlight band of its own — short tail sections that could + // never reach a fixed line are not skipped, and the highlight steps through sections + // in order. Scroll-position based rather than an IntersectionObserver: with an + // observer nothing intersects the narrow band between headings, so the highlight + // would stall while scrolling. + useEffect(() => { + if (toc.length < 2) return; + // Deep links carry the anchor percent-encoded (CJK headings); pin it if it is ours. + const initialHash = decodeURIComponent(window.location.hash.slice(1)); + pinnedId.current = toc.some((entry) => entry.id === initialHash) ? initialHash : null; + let raf = 0; + const update = () => { + raf = 0; + if (pinnedId.current !== null) { + setActiveId(pinnedId.current); + return; + } + const docEl = document.documentElement; + const maxScroll = Math.max(0, docEl.scrollHeight - window.innerHeight); + const progress = maxScroll > 0 ? Math.min(1, window.scrollY / maxScroll) : 0; + const line = 100 + Math.max(0, window.innerHeight - 140) * progress; + let current = toc[0]!.id; + for (const entry of toc) { + const el = document.getElementById(entry.id); + if (el && el.getBoundingClientRect().top <= line) current = entry.id; + } + // Safety net: fully at the bottom nothing can advance further — settle on the last + // entry (only after the page actually scrolled; a viewport-short page stays on top). + if (maxScroll > 0 && window.scrollY >= maxScroll - 4) current = toc[toc.length - 1]!.id; + setActiveId(current); + }; + const onScroll = () => { + if (raf === 0) raf = requestAnimationFrame(update); + }; + const unpin = (e?: KeyboardEvent) => { + if (e && !["ArrowDown", "ArrowUp", "PageDown", "PageUp", "Home", "End", " "].includes(e.key)) + return; + if (pinnedId.current === null) return; + pinnedId.current = null; + onScroll(); + }; + const onWheel = () => unpin(); + const onKey = (e: KeyboardEvent) => unpin(e); + // Same-page hash navigation (address bar / in-content anchors) re-runs no effect, + // so re-evaluate the pin whenever the hash changes. + const onHashChange = () => { + const hash = decodeURIComponent(window.location.hash.slice(1)); + if (toc.some((entry) => entry.id === hash)) { + pinnedId.current = hash; + setActiveId(hash); + } + }; + update(); + window.addEventListener("scroll", onScroll, { passive: true }); + window.addEventListener("resize", onScroll); + window.addEventListener("wheel", onWheel, { passive: true }); + window.addEventListener("touchmove", onWheel, { passive: true }); + window.addEventListener("keydown", onKey); + window.addEventListener("hashchange", onHashChange); + return () => { + window.removeEventListener("scroll", onScroll); + window.removeEventListener("resize", onScroll); + window.removeEventListener("wheel", onWheel); + window.removeEventListener("touchmove", onWheel); + window.removeEventListener("keydown", onKey); + window.removeEventListener("hashchange", onHashChange); + if (raf) cancelAnimationFrame(raf); + }; + }, [toc]); + + if (!doc) { + return ( +
+

{S.doc.notFound}

+ + + {S.doc.backHome} + +
+ ); + } + + const showToc = toc.length >= 2; + + return ( +
+
+
+
+
+ {section && ( +

+ {S.sections[section.id]} +

+ )} +

+ {doc.title} +

+
+
+ +
+
+ {doc.description && ( +

{doc.description}

+ )} +
+
+ {children}, + h2: ({ children }) => ( +

+ {children} +

+ ), + h3: ({ children }) => ( +

+ {children} +

+ ), + }} + > + {doc.body} +
+
+ +
+ {showToc && } +
+ ); +} diff --git a/packages/docs/src/router.tsx b/packages/docs/src/router.tsx new file mode 100644 index 0000000..f0c8b94 --- /dev/null +++ b/packages/docs/src/router.tsx @@ -0,0 +1,81 @@ +/** + * Router + layout: sticky top bar, left sidebar (sticky column on desktop, overlay + * panel on mobile), doc content, slim footer. basename comes from Vite's BASE_URL so + * the site works under the GitHub Pages subpath ("//docs/"); scroll restores to + * top on route change (hash targets excluded). + */ +import { useEffect, useState } from "react"; +import { BrowserRouter, Outlet, Route, Routes, useLocation } from "react-router"; +import { Nav } from "./components/nav"; +import { Sidebar } from "./components/sidebar"; +import { Footer } from "./components/footer"; +import { DocPage } from "./pages/doc-page"; + +/** + * Last history entry whose scroll was already handled. Module-level so it survives + * the locale-keyed remount of the whole tree: switching language re-mounts Layout, + * and without this guard the navigation effect would re-run (jumping to the hash or + * to the top) and defeat LocaleScope's scroll preservation. + */ +let handledLocationKey = ""; + +function Layout() { + const { pathname, hash, key } = useLocation(); + const [menuOpen, setMenuOpen] = useState(false); + + // Genuine route change: jump to top; with a hash scroll to the target once it is + // in the DOM. Also close the mobile sidebar. + useEffect(() => { + setMenuOpen(false); + if (key === handledLocationKey) return; + handledLocationKey = key; + if (hash) { + // Anchors of CJK headings arrive percent-encoded in the URL hash. + const el = document.getElementById(decodeURIComponent(hash.slice(1))); + if (el) { + el.scrollIntoView(); + return; + } + } + window.scrollTo(0, 0); + }, [pathname, hash, key]); + + return ( +
+
+ ); +} + +export function AppRouter() { + const basename = import.meta.env.BASE_URL.replace(/\/+$/, "") || "/"; + return ( + + + }> + } /> + } /> + + + + ); +} diff --git a/packages/docs/src/state/locale.tsx b/packages/docs/src/state/locale.tsx new file mode 100644 index 0000000..085292c --- /dev/null +++ b/packages/docs/src/state/locale.tsx @@ -0,0 +1,114 @@ +/** + * Language context: zh / en / system (tracks navigator.language). On switch it first + * synchronously calls setActiveStrings, then remounts the tree keyed on locale so every + * `S.x` read reflects the new language; the preference persists to localStorage. + * Same pattern as the landing page, under the docs site's own storage key. + */ +import { + createContext, + useCallback, + useContext, + useEffect, + useLayoutEffect, + useRef, + useState, +} from "react"; +import type { ReactNode } from "react"; +import { setActiveStrings, zh } from "../lib/strings"; +import { en } from "../lib/strings-en"; + +export type LangPref = "zh" | "en" | "system"; +export type Locale = "zh" | "en"; + +const STORAGE_KEY = "penguin-docs.lang"; + +interface LocaleContextValue { + lang: LangPref; + locale: Locale; + setLang: (lang: LangPref) => void; +} + +const LocaleContext = createContext(null); + +/** Device language -> UI language: zh* -> zh, anything else -> en. */ +export function resolveSystemLocale(language: string | undefined): Locale { + return language?.toLowerCase().startsWith("zh") ? "zh" : "en"; +} + +function systemLocale(): Locale { + return resolveSystemLocale(navigator.language); +} + +function resolve(lang: LangPref): Locale { + return lang === "system" ? systemLocale() : lang; +} + +function initialLang(): LangPref { + const stored = localStorage.getItem(STORAGE_KEY); + if (stored === "zh" || stored === "en" || stored === "system") return stored; + return "system"; +} + +export function LocaleProvider({ children }: { children: ReactNode }) { + const [lang, setLangState] = useState(initialLang); + const [, setSysTick] = useState(0); + + const locale = resolve(lang); + // Switch the active dictionary during render (idempotent): children are keyed on + // locale and render after this component, so they read the post-switch dictionary. + setActiveStrings(locale === "en" ? en : zh); + + // Keep the document language in sync (static index.html ships lang="en"). + useEffect(() => { + document.documentElement.lang = locale === "zh" ? "zh-CN" : "en"; + }, [locale]); + + useEffect(() => { + if (lang !== "system") return; + const onChange = () => setSysTick((t) => t + 1); + window.addEventListener("languagechange", onChange); + return () => window.removeEventListener("languagechange", onChange); + }, [lang]); + + const setLang = useCallback((next: LangPref) => { + localStorage.setItem(STORAGE_KEY, next); + setLangState(next); + }, []); + + return ( + {children} + ); +} + +/** + * Language scope: a remount boundary keyed on locale. The remount briefly empties the + * DOM, which collapses the page height and clamps the scroll position to 0 — so the + * scroll offset is captured during the render that switches locale (old DOM still + * mounted) and restored right after the new tree lays out. + */ +export function LocaleScope({ children }: { children: ReactNode }) { + const { locale } = useLocale(); + const prevLocale = useRef(locale); + const savedScroll = useRef(null); + if (prevLocale.current !== locale) { + prevLocale.current = locale; + savedScroll.current = window.scrollY; + } + useLayoutEffect(() => { + if (savedScroll.current !== null) { + window.scrollTo({ top: savedScroll.current, behavior: "instant" }); + savedScroll.current = null; + } + }, [locale]); + return ( +
+ {children} +
+ ); +} + +export function useLocale(): LocaleContextValue { + const ctx = useContext(LocaleContext); + if (!ctx) throw new Error("useLocale must be used inside LocaleProvider"); + return ctx; +} diff --git a/packages/docs/src/state/theme.tsx b/packages/docs/src/state/theme.tsx new file mode 100644 index 0000000..d72828c --- /dev/null +++ b/packages/docs/src/state/theme.tsx @@ -0,0 +1,64 @@ +/** + * Theme context: light / dark / system (tracks prefers-color-scheme live), toggled + * via the html.dark class + Tailwind dark: variant, persisted to localStorage. + * Same behavior as the landing page, under the docs site's own storage key + * (index.html pre-applies the stored value before first paint). + */ +import { createContext, useCallback, useContext, useEffect, useState } from "react"; +import type { ReactNode } from "react"; + +export type ThemeMode = "light" | "dark" | "system"; + +const MODE_KEY = "penguin-docs.theme"; + +interface ThemeContextValue { + mode: ThemeMode; + /** Resolved effective theme (system mode resolved against the OS preference). */ + dark: boolean; + setMode: (mode: ThemeMode) => void; +} + +const ThemeContext = createContext(null); + +function initialMode(): ThemeMode { + const stored = localStorage.getItem(MODE_KEY); + if (stored === "light" || stored === "dark" || stored === "system") return stored; + return "system"; +} + +function systemDark(): boolean { + return window.matchMedia("(prefers-color-scheme: dark)").matches; +} + +export function ThemeProvider({ children }: { children: ReactNode }) { + const [mode, setModeState] = useState(initialMode); + const [sysDark, setSysDark] = useState(systemDark); + + const dark = mode === "system" ? sysDark : mode === "dark"; + + useEffect(() => { + document.documentElement.classList.toggle("dark", dark); + }, [dark]); + + useEffect(() => { + if (mode !== "system") return; + const mq = window.matchMedia("(prefers-color-scheme: dark)"); + const onChange = (e: MediaQueryListEvent) => setSysDark(e.matches); + setSysDark(mq.matches); + mq.addEventListener("change", onChange); + return () => mq.removeEventListener("change", onChange); + }, [mode]); + + const setMode = useCallback((next: ThemeMode) => { + localStorage.setItem(MODE_KEY, next); + setModeState(next); + }, []); + + return {children}; +} + +export function useTheme(): ThemeContextValue { + const ctx = useContext(ThemeContext); + if (!ctx) throw new Error("useTheme must be used inside ThemeProvider"); + return ctx; +} diff --git a/packages/docs/src/styles.css b/packages/docs/src/styles.css new file mode 100644 index 0000000..050ba22 --- /dev/null +++ b/packages/docs/src/styles.css @@ -0,0 +1,191 @@ +/** + * Docs site styles: single-file Tailwind CSS 4 entry point sharing the landing page's + * visual language (GitHub-style simplicity, solid backgrounds + 1px borders + one + * brand blue accent) minus its marketing effects — doc pages stay calm and readable. + * Dark theme is toggled via the html.dark class. + */ +@import "tailwindcss"; + +/* Dark theme toggled via the html.dark class (Tailwind 4 custom variant). */ +@custom-variant dark (&:where(.dark, .dark *)); + +/* Brand color scale (Google blue family), identical to the landing page / Web App. */ +@theme { + --color-brand-25: #f8fbff; + --color-brand-50: #e8f0fe; + --color-brand-100: #d2e3fc; + --color-brand-200: #aecbfa; + --color-brand-300: #8ab4f8; + --color-brand-400: #669df6; + --color-brand-500: #4285f4; + --color-brand-600: #1a73e8; + --color-brand-700: #0b57d0; + --color-brand-800: #0842a0; + --color-brand-900: #062e6f; + --color-brand-950: #041e49; +} + +@layer base { + :root { + color-scheme: light; + } + .dark { + color-scheme: dark; + /* Pure-black dark base, matching the landing page (true-neutral gray overrides). */ + --color-gray-950: #000000; + --color-gray-900: #0d0d0d; + --color-gray-800: #1f1f1f; + --color-gray-700: #303030; + } + html { + scroll-behavior: smooth; + } + body { + @apply bg-white text-gray-900 antialiased dark:bg-gray-950 dark:text-gray-100; + font-family: + ui-sans-serif, + system-ui, + -apple-system, + "Segoe UI", + Roboto, + "PingFang SC", + "Microsoft YaHei", + sans-serif; + } + code, + pre, + kbd { + font-family: ui-monospace, SFMono-Regular, Menlo, Consolas, "Liberation Mono", monospace; + } + button:focus-visible, + a:focus-visible, + summary:focus-visible { + outline: 3px solid rgb(107 114 128 / 0.4); + outline-offset: 2px; + } + button:not(:disabled), + [role="button"]:not(:disabled) { + cursor: pointer; + } + ::selection { + background: rgb(0 0 0 / 0.1); + } + .dark ::selection { + background: rgb(255 255 255 / 0.18); + } + * { + scrollbar-width: thin; + scrollbar-color: rgb(60 64 67 / 0.28) transparent; + } + .dark * { + scrollbar-color: rgb(232 234 237 / 0.2) transparent; + } +} + +/* ---------- Animations (same tone as the landing page: short ease, slight offset) ---------- */ + +@keyframes rise-in { + from { + opacity: 0; + transform: translateY(10px) scale(0.99); + } + to { + opacity: 1; + transform: none; + } +} + +@keyframes fade-in { + from { + opacity: 0; + } + to { + opacity: 1; + } +} + +.anim-rise { + animation: rise-in 280ms cubic-bezier(0.2, 0.7, 0.3, 1) both; +} +.anim-fade { + animation: fade-in 120ms ease-out both; +} + +@media (prefers-reduced-motion: reduce) { + *, + *::before, + *::after { + animation: none !important; + transition: none !important; + scroll-behavior: auto !important; + } +} + +/* ---------- Typography for Markdown doc bodies (same voice as the landing blog) ---------- */ + +.md-body { + overflow-wrap: break-word; +} +.md-body :is(p, ul, ol, pre, blockquote, table) { + margin: 0.625rem 0; +} +.md-body li { + margin: 0.3rem 0; +} +.md-body p, +.md-body li { + line-height: 1.75; +} +.md-body :is(h1, h2, h3, h4) { + font-weight: 600; + margin: 1.75rem 0 0.5rem; +} +.md-body h1 { + font-size: 1.375rem; +} +.md-body h2 { + font-size: 1.1875rem; + @apply border-b border-gray-200 pb-1.5 dark:border-gray-800; +} +.md-body h3 { + font-size: 1.0625rem; +} +.md-body ul { + list-style: disc; + padding-left: 1.25rem; +} +.md-body ol { + list-style: decimal; + padding-left: 1.25rem; +} +.md-body a { + @apply text-brand-700 underline decoration-brand-300 underline-offset-2 transition-colors hover:text-brand-600 dark:text-brand-300 dark:decoration-brand-700; +} +.md-body code { + @apply rounded bg-gray-100 px-1 py-0.5 text-[0.85em] text-gray-800 dark:bg-gray-800 dark:text-gray-200; +} +.md-body pre { + @apply overflow-x-auto rounded-lg border border-gray-200 bg-gray-50 p-3 text-[13px] leading-6 dark:border-gray-800 dark:bg-gray-900; +} +.md-body pre code { + background: transparent; + color: inherit; + padding: 0; +} +.md-body blockquote { + @apply border-l-2 border-gray-300 pl-3 text-gray-600 dark:border-gray-700 dark:text-gray-400; +} +.md-body table { + border-collapse: collapse; + display: block; + overflow-x: auto; +} +.md-body :is(th, td) { + @apply border border-gray-200 px-2.5 py-1.5 text-sm dark:border-gray-800; +} +.md-body th { + @apply bg-gray-50 text-left dark:bg-gray-900; +} +.md-body hr { + @apply my-4 border-gray-200 dark:border-gray-800; +} diff --git a/packages/docs/test/content.test.ts b/packages/docs/test/content.test.ts new file mode 100644 index 0000000..9d5462b --- /dev/null +++ b/packages/docs/test/content.test.ts @@ -0,0 +1,41 @@ +/** + * Content integrity: the sidebar (DOCS_NAV) and the content/ directory must agree — + * every navigated slug has both zh and en files with a frontmatter title, and every + * content file belongs to the navigation (an orphan file would be unreachable). + * Reads the files via fs so the check runs under plain node (no Vite glob). + */ +import { readdirSync, readFileSync } from "node:fs"; +import { join } from "node:path"; +import { describe, expect, it } from "vitest"; +import { DOCS_NAV, DOC_SLUGS } from "../src/lib/nav"; +import { parseFrontmatter } from "../src/lib/frontmatter"; + +const contentDir = join(__dirname, "..", "content"); +const files = readdirSync(contentDir).filter((f) => f.endsWith(".md")); + +describe("docs navigation / content integrity", () => { + it("has unique slugs in DOCS_NAV", () => { + expect(new Set(DOC_SLUGS).size).toBe(DOC_SLUGS.length); + expect(DOCS_NAV.length).toBeGreaterThan(0); + }); + + it("provides zh and en files with a title for every navigated slug", () => { + for (const slug of DOC_SLUGS) { + for (const lang of ["zh", "en"] as const) { + const name = `${slug}.${lang}.md`; + expect(files, `missing content file ${name}`).toContain(name); + const { meta, body } = parseFrontmatter(readFileSync(join(contentDir, name), "utf8")); + expect(meta.title, `missing title in ${name}`).toBeTruthy(); + expect(body.length, `empty body in ${name}`).toBeGreaterThan(0); + } + } + }); + + it("has no content file outside the navigation", () => { + for (const file of files) { + const slug = /^(.+)\.(zh|en)\.md$/.exec(file)?.[1]; + expect(slug, `unparsable content file name ${file}`).toBeTruthy(); + expect(DOC_SLUGS, `orphan content file ${file}`).toContain(slug!); + } + }); +}); diff --git a/packages/docs/test/frontmatter.test.ts b/packages/docs/test/frontmatter.test.ts new file mode 100644 index 0000000..40983b7 --- /dev/null +++ b/packages/docs/test/frontmatter.test.ts @@ -0,0 +1,25 @@ +import { describe, expect, it } from "vitest"; +import { parseFrontmatter } from "../src/lib/frontmatter"; + +describe("parseFrontmatter", () => { + it("parses a leading key/value block and trims the body", () => { + const { meta, body } = parseFrontmatter( + '---\ntitle: "OmniMessage"\ndescription: one protocol: three jobs\n---\n\nBody text\n', + ); + expect(meta.title).toBe("OmniMessage"); + expect(meta.description).toBe("one protocol: three jobs"); + expect(body).toBe("Body text"); + }); + + it("returns the whole input as body when there is no frontmatter", () => { + const { meta, body } = parseFrontmatter("# Just markdown\n"); + expect(meta).toEqual({}); + expect(body).toBe("# Just markdown"); + }); + + it("normalizes CRLF line endings", () => { + const { meta, body } = parseFrontmatter("---\r\ntitle: X\r\n---\r\nbody\r\n"); + expect(meta.title).toBe("X"); + expect(body).toBe("body"); + }); +}); diff --git a/packages/docs/test/skills-sync.test.ts b/packages/docs/test/skills-sync.test.ts new file mode 100644 index 0000000..ad3b410 --- /dev/null +++ b/packages/docs/test/skills-sync.test.ts @@ -0,0 +1,31 @@ +/** + * Docs ↔ skill-library sync: the Skills doc pages must mention every Skill that + * actually ships in packages/skills (the library directory is the source of truth — + * the same files loadLibrarySkills() reads). Derived, not hardcoded, so adding a + * Skill without documenting it fails here instead of silently drifting. + */ +import { existsSync, readdirSync, readFileSync } from "node:fs"; +import { join } from "node:path"; +import { describe, expect, it } from "vitest"; + +const skillsRoot = join(__dirname, "..", "..", "skills", "skills"); +const contentDir = join(__dirname, "..", "content"); + +const librarySkills = readdirSync(skillsRoot, { withFileTypes: true }) + .filter((entry) => entry.isDirectory() && existsSync(join(skillsRoot, entry.name, "SKILL.md"))) + .map((entry) => entry.name) + .sort(); + +describe("docs ↔ skill library sync", () => { + it("found the skill library", () => { + expect(librarySkills.length).toBeGreaterThan(0); + }); + + for (const lang of ["zh", "en"] as const) { + it(`skills.${lang}.md mentions every library Skill`, () => { + const page = readFileSync(join(contentDir, `skills.${lang}.md`), "utf8"); + const missing = librarySkills.filter((name) => !page.includes(`\`${name}\``)); + expect(missing, `undocumented skills in skills.${lang}.md`).toEqual([]); + }); + } +}); diff --git a/packages/docs/test/toc.test.ts b/packages/docs/test/toc.test.ts new file mode 100644 index 0000000..1bf78f7 --- /dev/null +++ b/packages/docs/test/toc.test.ts @@ -0,0 +1,27 @@ +import { describe, expect, it } from "vitest"; +import { extractToc, slugifyHeading } from "../src/lib/toc"; + +describe("slugifyHeading", () => { + it("lowercases latin, hyphenates spaces, keeps CJK", () => { + expect(slugifyHeading("Agent Loop")).toBe("agent-loop"); + expect(slugifyHeading("消息信封")).toBe("消息信封"); + expect(slugifyHeading("Tool Calls (streaming)")).toBe("tool-calls-streaming"); + }); +}); + +describe("extractToc", () => { + it("collects ##/### headings and skips fenced code blocks", () => { + const body = [ + "## Envelope", + "```ts", + "## not a heading", + "```", + "### Payload kinds", + "#### too deep", + ].join("\n"); + expect(extractToc(body)).toEqual([ + { id: "envelope", text: "Envelope", depth: 2 }, + { id: "payload-kinds", text: "Payload kinds", depth: 3 }, + ]); + }); +}); diff --git a/packages/docs/tsconfig.json b/packages/docs/tsconfig.json new file mode 100644 index 0000000..0038d40 --- /dev/null +++ b/packages/docs/tsconfig.json @@ -0,0 +1,10 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { + "lib": ["ES2023", "DOM", "DOM.Iterable"], + "types": ["vite/client"], + "jsx": "react-jsx", + "useDefineForClassFields": true + }, + "include": ["src", "test", "vite.config.ts", "vitest.config.ts"] +} diff --git a/packages/docs/vite.config.ts b/packages/docs/vite.config.ts new file mode 100644 index 0000000..b424e81 --- /dev/null +++ b/packages/docs/vite.config.ts @@ -0,0 +1,19 @@ +/** + * Vite config: static docs site (React SPA + Tailwind CSS 4), a sibling of the + * landing page but its own package so either site can evolve independently. + * + * BASE_PATH is injected by the GitHub Pages workflow as "//docs/" — the docs + * build is copied into the landing dist under docs/ so both sites ship as one Pages + * artifact (see scripts/build-site.mjs at the repo root). Local dev defaults to "/". + * Doc pages are local Markdown files imported at build time via import.meta.glob + * (?raw), so the built site is fully static — no server or CMS involved. + */ +import react from "@vitejs/plugin-react"; +import tailwindcss from "@tailwindcss/vite"; +import { defineConfig } from "vite"; + +export default defineConfig({ + base: process.env.BASE_PATH ?? "/", + plugins: [react(), tailwindcss()], + server: { port: 7367 }, +}); diff --git a/packages/docs/vitest.config.ts b/packages/docs/vitest.config.ts new file mode 100644 index 0000000..e12ad25 --- /dev/null +++ b/packages/docs/vitest.config.ts @@ -0,0 +1,11 @@ +/** + * Vitest config kept separate from vite.config.ts (same convention as the landing + * package: vitest's embedded vite types conflict with this package's vite 7 plugin + * types). Tests cover pure modules and content integrity only, so a node + * environment with no plugins suffices. + */ +import { defineConfig } from "vitest/config"; + +export default defineConfig({ + test: { environment: "node" }, +}); diff --git a/packages/landing/README.md b/packages/landing/README.md new file mode 100644 index 0000000..640ebbd --- /dev/null +++ b/packages/landing/README.md @@ -0,0 +1,42 @@ +# @prismshadow/penguin-landing + +The PenguinHarness product landing page (React + Vite + Tailwind CSS 4): bilingual (zh/en), light/dark themes, the feature matrix and CONTRACT.md showcase, benchmark charts, Playwright-captured product screenshots, and a local-Markdown blog (product news / release notes). + +It deploys to GitHub Pages together with the [docs site](../docs) as one artifact: landing at the site root, docs under `/docs/` (assembled by `scripts/build-site.mjs` at the repo root; the nav, footer and CTA link into `/docs/`). + +## Development + +```bash +pnpm --filter @prismshadow/penguin-landing dev # http://127.0.0.1:7366 +pnpm --filter @prismshadow/penguin-landing build # dist/ (with 404.html SPA fallback + .nojekyll) +pnpm --filter @prismshadow/penguin-landing typecheck +pnpm --filter @prismshadow/penguin-landing test + +BASE_PATH=/ pnpm build:site # repo root: assemble landing + docs +pnpm --filter @prismshadow/penguin-landing preview # serve the assembled tree +``` + +`BASE_PATH` controls the asset base (GitHub Pages project pages need `//`; local default `/`). The `Docs` links resolve only in the assembled build — in plain `dev` they point at a path this dev server doesn't serve. + +## Deployment + +`.github/workflows/pages.yml` builds landing + docs and deploys on pushes to main that touch either package (or on manual dispatch). First-time setup: repository Settings → Pages → Source = "GitHub Actions". + +## Blog + +Posts live in `content/blog/..md` (`lang` is `zh` / `en`; the two language versions of a slug fall back to each other). Frontmatter: + +```markdown +--- +title: Post title +date: 2026-07-17 +category: news | changelog +excerpt: One-line summary for the list page +--- +``` + +## Screenshots + +`pnpm --filter @prismshadow/penguin-landing shots` regenerates `src/assets/shots/`: the script ships a scripted mock LLM (speaking both Anthropic SSE and OpenAI chat-completions streams), boots the Web service on a temp data root, drives a real "build an Agent app" conversation (tools actually execute in the Workspace), then captures the chat, trace and evaluation pages per UI language (zh/en, separate users) and theme (light/dark) — 12 shots (`--.webp`, re-encoded to WebP in Chromium to keep sizes down). The landing page always shows the shot matching the visitor's language and theme. Prereqs: build skills/core/server/web first, and have Playwright Chromium installed. + +Part of [PenguinHarness](https://github.com/Prism-Shadow/penguin-harness) · Apache-2.0 diff --git a/packages/landing/content/blog/introducing-penguinharness.en.md b/packages/landing/content/blog/introducing-penguinharness.en.md new file mode 100644 index 0000000..5816b8c --- /dev/null +++ b/packages/landing/content/blog/introducing-penguinharness.en.md @@ -0,0 +1,71 @@ +--- +title: "Introducing PenguinHarness: agents that build agents" +date: 2026-07-17 +category: news +excerpt: The first open-source harness with recursive self-improvement is here — lightweight, efficient and secure infrastructure covering everything from automatic agent construction to continuous self-evolution. +--- + +Today we are releasing **PenguinHarness** — an open-source harness built for constructing and evolving agents. Its purpose fits in one line: + +> Efficient Self-Improving Harness for Everyone. + +## Why PenguinHarness + +Over the past year the way agent applications are built has been converging fast: what really decides quality is not a heavyweight framework but a simple, reliable, observable harness. PenguinHarness is rebuilt from the ground up — no dependency on any agent framework, a fully open-source self-developed kernel — and it brings three things to the open-source world first: + +- **Simplest Is the Best**: a deliberately minimal toolset over clean low-level interfaces — fewer tool calls, fewer Tokens, complex tasks done efficiently. +- **Harness for Building Agents**: with the PenguinHarness SDK, an Agent builds complete Agent applications for you, autonomously, from scratch. +- **Harness for Recursive Self-Improvement**: with PenguinHarness Skills, an Agent evaluates and optimizes itself, improving recursively over time. + +For the latter two, PenguinHarness is the first open-source implementation in the industry. + +## Same model, equal or better quality, lower cost + +All runs use the same DeepSeek V4 Pro model, head-to-head against Claude Code and OpenAI Codex on two suites (per-run means below). + +Complex data analysis (15 tasks, single run): + +| Framework | Model | Accuracy (%) | Tokens (M) | Cost ($) | +| --- | --- | ---: | ---: | ---: | +| PenguinHarness | DeepSeek V4 Pro | 66.7 | 18.04 | 0.552 | +| Claude Code | DeepSeek V4 Pro | 66.7 | 21.17 | 0.641 | +| OpenAI Codex | DeepSeek V4 Pro | 46.7 | 13.36 | 0.427 | + +Coding tasks (40 tasks × 2 runs averaged, thinking high, 30 min per-case timeout, CNY pricing converted at $1 = ¥7): + +| Framework | Model | Accuracy (%) | Tokens (M) | Cost ($) | +| --- | --- | ---: | ---: | ---: | +| PenguinHarness | DeepSeek V4 Pro | 50.00 | 2.10 | 0.041 | +| Claude Code | DeepSeek V4 Pro | 48.75 | 2.00 | 0.048 | +| OpenAI Codex | DeepSeek V4 Pro | 42.50 | 2.65 | 0.043 | + +On the data-analysis suite PenguinHarness ties Claude Code on accuracy and clearly beats OpenAI Codex while using 14.8% fewer Tokens at 13.8% lower cost; on the coding suite it scores highest of the three at the lowest per-run cost. + +## Evolution within bounds, security first + +The biggest worry about self-improvement is losing control. PenguinHarness answers with a contract — CONTRACT.md: + +- Evolution is strictly confined to Workspace and Skills; the harness core security boundary is never modified. +- Tool calls run only after approval, and every approval is audited. +- Risky changes snapshot first — every step of evolution can be rolled back. +- Fully open source and locally deployable: data never leaves your machine, meeting enterprise security requirements. + +## Get started now + +Install with one command (Linux / macOS, x64 / arm64, bundled Node runtime): + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +``` + +Configure a model (DeepSeek as an example), run your first task, or open the desktop-grade web interface with `penguin web`: + +```bash +penguin config model add --model-id deepseek-v4-pro --api-key sk-your-key --set-default +penguin run --approve allow-all --message "Analyze data.csv and summarize quarterly sales" +penguin web +``` + +PenguinHarness supports 1000+ online and local models and multi-agent collaborative evolution, and runs on as little as a single CPU. Through continuous evolution it makes complex AI development ever simpler — a more efficient, more reliable, lower-hallucination and lower-cost Agent productivity engine. + +Follow us on [GitHub](https://github.com/Prism-Shadow/penguin-harness) and open your first issue. diff --git a/packages/landing/content/blog/introducing-penguinharness.zh.md b/packages/landing/content/blog/introducing-penguinharness.zh.md new file mode 100644 index 0000000..6b34c4d --- /dev/null +++ b/packages/landing/content/blog/introducing-penguinharness.zh.md @@ -0,0 +1,71 @@ +--- +title: PenguinHarness 正式发布:让 Agent 为你构建 Agent +date: 2026-07-17 +category: news +excerpt: 首个支持递归自我进化的开源 Harness 正式发布——以轻量、高效、安全的方式,提供从 Agent 自动构建到持续自我进化的完整基础设施。 +--- + +今天,我们正式发布 **PenguinHarness**——一个为构建与进化 Agent 而生的开源 Harness。它的主旨只有一句话: + +> Efficient Self-Improving Harness for Everyone. + +## 为什么是 PenguinHarness + +过去一年里,Agent 应用的开发范式在快速收敛:真正决定效果的不是庞大的框架,而是一个简洁、可靠、可观测的 Harness。PenguinHarness 从底层重构,不依赖任何 Agent 框架,开源自研 Harness 内核,并率先把三件事带入开源世界: + +- **Simplest Is the Best**:坚持最小化工具集与简洁的底层接口,以更少的工具调用与 Token 消耗,高效完成复杂任务。 +- **Harness for Building Agents**:通过 PenguinHarness SDK,让 Agent 从零自主完成 Agent 应用的构建。 +- **Harness for Recursive Self-Improvement**:通过 PenguinHarness Skills,Agent 以自我评估与自我优化实现递归式自我提升。 + +后两项能力,PenguinHarness 是业内首个开源实现。 + +## 同一模型,同级效果,更低消耗 + +全部使用同一 DeepSeek V4 Pro 模型,与 Claude Code、OpenAI Codex 在两套题库上正面对比(表中为单次运行均值)。 + +复杂数据分析(15 题,单次运行): + +| 实验框架 | 模型名称 | 准确率(%) | Token 用量(M) | 成本($) | +| --- | --- | ---: | ---: | ---: | +| PenguinHarness | DeepSeek V4 Pro | 66.7 | 18.04 | 0.552 | +| Claude Code | DeepSeek V4 Pro | 66.7 | 21.17 | 0.641 | +| OpenAI Codex | DeepSeek V4 Pro | 46.7 | 13.36 | 0.427 | + +代码任务(40 题 × 2 runs 取均值,thinking high、单题 30 分钟超时,人民币计价按 $1 = ¥7 折算): + +| 实验框架 | 模型名称 | 准确率(%) | Token 用量(M) | 成本($) | +| --- | --- | ---: | ---: | ---: | +| PenguinHarness | DeepSeek V4 Pro | 50.00 | 2.10 | 0.041 | +| Claude Code | DeepSeek V4 Pro | 48.75 | 2.00 | 0.048 | +| OpenAI Codex | DeepSeek V4 Pro | 42.50 | 2.65 | 0.043 | + +数据分析套件与 Claude Code 准确率持平、显著超过 OpenAI Codex,同时 Token 消耗少 14.8%、成本低 13.8%;代码套件三者中准确率最高、单次成本最低。 + +## 进化有界,安全先行 + +自我进化最大的疑虑是失控。PenguinHarness 用一份契约(CONTRACT.md)回答这个问题: + +- 进化严格限制在 Workspace 与 Skill 之内,不修改 Harness 核心安全边界; +- 工具调用先经批准,每次批准皆留审计; +- 风险修改之前先留版本快照,任何一次进化都可回退; +- 完全开源、本地部署,数据不出域,满足企业级数据安全。 + +## 现在就可以开始 + +一行命令安装(Linux / macOS,x64 / arm64,内嵌 Node 运行时): + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +``` + +配置模型(以 DeepSeek 为例)后即可运行第一个任务,或用 `penguin web` 打开桌面级 Web 界面: + +```bash +penguin config model add --model-id deepseek-v4-pro --api-key sk-your-key --set-default +penguin run --approve allow-all --message "分析 data.csv,输出各季度销售额汇总" +penguin web +``` + +PenguinHarness 支持 1000 多种在线与本地模型、多智能体协作进化,最低单 CPU 即可运行。通过不断进化,它会让复杂的 AI 开发越来越简单——为你提供更高效、更可靠、更低幻觉、更低成本的 Agent 生产力引擎。 + +欢迎在 [GitHub](https://github.com/Prism-Shadow/penguin-harness) 上关注我们,提出你的第一个 Issue。 diff --git a/packages/landing/content/blog/july-2026-updates.en.md b/packages/landing/content/blog/july-2026-updates.en.md new file mode 100644 index 0000000..10053dc --- /dev/null +++ b/packages/landing/content/blog/july-2026-updates.en.md @@ -0,0 +1,30 @@ +--- +title: "July 2026 updates: scheduled tasks, Agent snapshots and a stronger evaluation center" +date: 2026-07-17 +category: changelog +excerpt: Scheduled tasks, Agent State snapshots with export/import, benchmark scoreboards, the model identity principle and one-line install have all landed on main. +--- + +This month a batch of updates directly serving "stable evolution" landed on main. Highlights below. + +## Scheduled tasks & Agent State snapshots + +- **Scheduled tasks**: one TOML file per task under `agent_state/schedule/` — cron-style scheduling keeps Agents working autonomously around the clock. +- **Agent State snapshots with export/import**: `system_config.yaml` carries a `version`; risky changes (optimization passes, import overwrite) snapshot to `snapshots/v.tar.gz` first, restore any time, with the live vault preserved. + +## Evaluation center + +- **Benchmark scoreboards**: bundled suites, per-case scoring and trend curves; evaluations are charted per model, and each run deep-links to its Session's trace view. +- **Evaluations carry the model**: the model reference moved from benchmark_config onto each evaluation (`provider` / `model_id` as a pair), making cross-model comparison direct. + +## Model system + +- **Model identity principle**: a model is uniquely identified by the `(provider, model_id)` pair; connection details live inline on the Project config entry, with client-resolved environment-variable fallback when the credential is left empty. +- **Custom provider groups**: beyond built-in vendors and custom, users can create their own groups (OpenAI protocol by default, base URL required). +- **Runtime baseline raised to Node ≥ 24**: bundled runtime, CI and the release pipeline all migrated. + +## Install & experience + +- **One-line install**: a repo-root `install.sh` — `curl | sh` detects Linux / macOS and x64 / arm64; artifacts bundle the Node runtime, unpack and run. +- **Skill library revamp**: Skills now use files as the runtime source of truth, with redesigned cards and quick invocation; built-in Agents converge to a single default_agent, with agent creation/optimization fully carried by the Skill library. +- **Stability fixes**: Gemini tool_call_id collisions on consecutive same-name calls, stream-view scroll jitter on short containers, and WorkGroup parallel tool timing are all fixed. diff --git a/packages/landing/content/blog/july-2026-updates.zh.md b/packages/landing/content/blog/july-2026-updates.zh.md new file mode 100644 index 0000000..8d12415 --- /dev/null +++ b/packages/landing/content/blog/july-2026-updates.zh.md @@ -0,0 +1,30 @@ +--- +title: 2026 年 7 月更新:定时任务、Agent 快照与评估中心增强 +date: 2026-07-17 +category: changelog +excerpt: 定时任务调度、Agent State 版本快照与导出导入、Benchmark 记分展示、模型确定原则与一键安装等一批更新已合入主干。 +--- + +本月主干合入了一批与「稳定进化」直接相关的更新,摘要如下。 + +## 定时任务与 Agent State 快照 + +- **定时任务调度**:`agent_state/schedule/` 下每个任务一个 TOML 文件,计划调度让 Agent 全天候自主执行。 +- **Agent State 版本快照与导出导入**:`system_config.yaml` 以 `version` 标识当前版本,风险修改(优化类修改、导入覆盖)前自动快照到 `snapshots/v.tar.gz`,可随时回退,且恢复时保留现行 vault。 + +## 评估中心增强 + +- **Benchmark 记分展示**:内建题库、逐题评分与趋势曲线,评估记录按模型分系列展示,run 可直达对应 Session 的轨迹观测。 +- **评估携带模型**:模型引用从 benchmark_config 移到每条 evaluation(`provider` / `model_id` 成对记录),跨模型对比更直观。 + +## 模型体系 + +- **模型确定原则**:模型由 `(provider, model_id)` 二元组唯一确定,连接信息内联在 Project 配置的模型条目上;credential 留空时按 client 解析结果回退环境变量。 +- **模型页自建分组**:内置厂商分组与 custom 之外,支持用户自建分组(OpenAI 协议缺省、base URL 必填)。 +- **运行基线升级 Node ≥ 24**:内嵌运行时同步更新,CI 与发布链路完成迁移。 + +## 安装与体验 + +- **一键安装**:仓库根新增 `install.sh`,`curl | sh` 识别 Linux / macOS 与 x64 / arm64,产物内嵌 Node 运行时,解压即用。 +- **技能库改版**:Skill 以文件为运行时真源,技能卡片改版并支持快捷调用;内置 Agent 收敛为 default_agent,Agent 构建与优化能力全部由技能库承载。 +- **稳定性修复**:Gemini 连续同名工具调用 tool_call_id 冲突、流式输出短滚动区上滑抖动、WorkGroup 并行工具用时统计等问题已修复。 diff --git a/packages/landing/index.html b/packages/landing/index.html new file mode 100644 index 0000000..18a5b49 --- /dev/null +++ b/packages/landing/index.html @@ -0,0 +1,26 @@ + + + + + + + + + + + + + + PenguinHarness — Efficient Self-Improving Harness for Everyone + + +
+ + + diff --git a/packages/landing/package.json b/packages/landing/package.json new file mode 100644 index 0000000..271c660 --- /dev/null +++ b/packages/landing/package.json @@ -0,0 +1,34 @@ +{ + "name": "@prismshadow/penguin-landing", + "version": "0.0.1", + "private": true, + "type": "module", + "description": "PenguinHarness Landing Page:产品落地页(React + Vite + Tailwind CSS),多语言、亮暗主题、Benchmark 展示与本地 Markdown 博客,经 GitHub Actions 部署到 GitHub Pages。", + "scripts": { + "dev": "vite", + "build": "vite build && node scripts/postbuild.mjs", + "preview": "vite preview", + "typecheck": "tsc --noEmit -p tsconfig.json", + "test": "vitest run --passWithNoTests", + "shots": "node scripts/capture-shots.mjs" + }, + "dependencies": { + "react": "^19.1.0", + "react-dom": "^19.1.0", + "react-markdown": "^10.1.0", + "react-router": "^7.6.0", + "remark-gfm": "^4.0.0" + }, + "devDependencies": { + "@playwright/test": "^1.61.1", + "@tailwindcss/vite": "^4.1.0", + "@types/node": "^24.0.0", + "@types/react": "^19.1.0", + "@types/react-dom": "^19.1.0", + "@vitejs/plugin-react": "^4.5.0", + "tailwindcss": "^4.1.0", + "typescript": "^5.6.0", + "vite": "^7.0.0", + "vitest": "^2.1.0" + } +} diff --git a/packages/landing/public/penguin-logo.svg b/packages/landing/public/penguin-logo.svg new file mode 100644 index 0000000..0d48619 --- /dev/null +++ b/packages/landing/public/penguin-logo.svg @@ -0,0 +1 @@ + diff --git a/packages/landing/scripts/capture-shots.mjs b/packages/landing/scripts/capture-shots.mjs new file mode 100644 index 0000000..56559f5 --- /dev/null +++ b/packages/landing/scripts/capture-shots.mjs @@ -0,0 +1,586 @@ +/** + * Capture real product screenshots for the landing page. + * + * Flow: host a scripted mock LLM (speaks BOTH Anthropic SSE and OpenAI chat-completions + * SSE, so whichever client AgentHub routes to gets a valid stream) -> start the Web + * server against a temp data root serving the built web dist -> drive a genuine + * "build an Agent app" conversation (tools actually execute in the workspace) -> + * screenshot chat / trace view / evaluation center, per UI language (zh / en, each + * with its own user so sidebars stay monolingual) and per theme (light / dark), into + * src/assets/shots/ as --.webp (12 files, re-encoded to WebP + * inside Chromium to keep the repo small). + * + * Prereqs: `pnpm --filter @prismshadow/penguin-{skills,core,server,web} build` and + * Playwright's chromium. Run: `node scripts/capture-shots.mjs`. + */ +import http from "node:http"; +import { spawn } from "node:child_process"; +import { mkdtempSync, mkdirSync, writeFileSync } from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { chromium } from "@playwright/test"; + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const ROOT = path.resolve(HERE, "../../.."); +const OUT_DIR = path.resolve(HERE, "../src/assets/shots"); +const MOCK_PORT = 8941; +const SRV_PORT = 8940; +const BASE = `http://127.0.0.1:${SRV_PORT}`; +const MOCK = `http://127.0.0.1:${MOCK_PORT}`; + +// --------------------------------------------------------------------------- +// Scripted conversation: the Agent builds an Agent application from scratch. +// Commands are shared across languages (code is code) and really execute. +// --------------------------------------------------------------------------- + +const CMD_SCAFFOLD = `mkdir -p csv-analyst/src && cat > csv-analyst/package.json <<'EOF' +{ + "name": "csv-analyst", + "private": true, + "type": "module", + "scripts": { "start": "tsx src/agent.ts" }, + "dependencies": { "@prismshadow/penguin-core": "^0.1.0" } +} +EOF +ls -R csv-analyst`; + +const CMD_ENTRY = `cat > csv-analyst/src/agent.ts <<'EOF' +import { createAgent, isCompleteModelMessage, userText } from "@prismshadow/penguin-core"; + +const agent = await createAgent({ agentId: "csv_analyst" }); +const session = await agent.createSession({ workspaceDir: process.cwd() }); + +for await (const out of session.run([userText("Analyze data.csv and write summary.md")], { + approve: async () => "allow", +})) { + if (isCompleteModelMessage(out) && out.payload.type === "text") { + console.log(out.payload.text); + } +} +EOF +wc -l csv-analyst/src/agent.ts`; + +const TREE = `\`\`\`text +csv-analyst/ +├── package.json +└── src/ + └── agent.ts +\`\`\``; + +/** Per-language script: user prompt marker -> turns + session title. */ +const SCRIPTS = { + zh: { + marker: "数据分析 Agent 应用", + prompt: "用 PenguinHarness SDK 创建一个数据分析 Agent 应用:读取 CSV 并输出汇总报告", + title: "构建数据分析 Agent 应用", + turns: [ + { + thinking: + "需求是基于 penguin-core 的数据分析 Agent 应用。先创建项目骨架:package.json 与源码目录。", + text: "我来创建应用骨架:", + cmd: CMD_SCAFFOLD, + }, + { + thinking: + "骨架已建好。入口代码用 createAgent + createSession,把 CSV 分析任务交给 session.run。", + text: "骨架就绪,写入 Agent 入口代码:创建 Session,把 CSV 分析任务交给 session.run 并流式输出。", + cmd: CMD_ENTRY, + }, + { + text: `数据分析 Agent 应用已创建完成: + +${TREE} + +- 入口 \`csv-analyst/src/agent.ts\`:创建 Agent 与 Session,任务经 \`session.run\` 流式执行,工具调用逐个审批; +- 运行方式:\`cd csv-analyst && npm install && npm start\`; +- 建议下一步:在评估中心为它配一组 CSV 任务 Benchmark,交给 Optimizer 持续优化。`, + }, + ], + }, + en: { + marker: "data-analysis Agent app", + prompt: + "Use the PenguinHarness SDK to create a data-analysis Agent app that reads CSV files and writes a summary report", + title: "Build a data-analysis Agent app", + turns: [ + { + thinking: + "They want a data-analysis Agent app on penguin-core. Start with the project skeleton: package.json plus the source directory.", + text: "Let me scaffold the app first:", + cmd: CMD_SCAFFOLD, + }, + { + thinking: + "Skeleton is in place. The entry uses createAgent + createSession and hands the CSV task to session.run.", + text: "Skeleton ready — now the Agent entry point: create a Session and hand the CSV analysis task to session.run, streaming the output.", + cmd: CMD_ENTRY, + }, + { + text: `The data-analysis Agent app is ready: + +${TREE} + +- Entry \`csv-analyst/src/agent.ts\`: creates the Agent and a Session; the task runs through \`session.run\` with per-tool approval; +- Run it with \`cd csv-analyst && npm install && npm start\`; +- Suggested next step: give it a CSV Benchmark suite in the evaluation center and let an Optimizer keep improving it.`, + }, + ], + }, +}; + +function scriptFor(flat) { + return flat.includes(SCRIPTS.en.marker) ? SCRIPTS.en : SCRIPTS.zh; +} + +// --------------------------------------------------------------------------- +// Mock LLM: Anthropic SSE on */messages, OpenAI chunks on */chat/completions. +// --------------------------------------------------------------------------- + +function sse(res, event, data) { + res.write(`event: ${event}\n`); + res.write(`data: ${JSON.stringify(data)}\n\n`); +} + +function anthropicReply(res, body) { + const flat = JSON.stringify(body.messages ?? []); + const script = scriptFor(flat); + const isTitle = flat.includes("concise title"); + const toolResults = flat.split('"tool_result"').length - 1; + const turn = script.turns[Math.min(toolResults, script.turns.length - 1)]; + const msgCount = (body.messages ?? []).length; + + res.writeHead(200, { "content-type": "text/event-stream", "cache-control": "no-cache" }); + sse(res, "message_start", { + type: "message_start", + message: { + id: `msg_shot_${Date.now()}`, + type: "message", + role: "assistant", + model: "deepseek-v4-pro", + content: [], + stop_reason: null, + stop_sequence: null, + usage: { + input_tokens: 380, + output_tokens: 0, + cache_read_input_tokens: 2400 * msgCount, + cache_creation_input_tokens: 620, + }, + }, + }); + + const block = (index, start, deltas, extra) => { + sse(res, "content_block_start", { type: "content_block_start", index, content_block: start }); + for (const d of deltas) + sse(res, "content_block_delta", { type: "content_block_delta", index, delta: d }); + if (extra) + sse(res, "content_block_delta", { type: "content_block_delta", index, delta: extra }); + sse(res, "content_block_stop", { type: "content_block_stop", index }); + }; + const finish = (stopReason, outputTokens) => { + sse(res, "message_delta", { + type: "message_delta", + delta: { stop_reason: stopReason, stop_sequence: null }, + usage: { output_tokens: outputTokens }, + }); + sse(res, "message_stop", { type: "message_stop" }); + res.end(); + }; + const textDeltas = (text) => + (text.match(/[\s\S]{1,24}/g) ?? []).map((t) => ({ type: "text_delta", text: t })); + + if (isTitle) { + block(0, { type: "text", text: "" }, [{ type: "text_delta", text: script.title }]); + finish("end_turn", 8); + return; + } + + let index = 0; + if (turn.thinking) { + block( + index++, + { type: "thinking", thinking: "" }, + turn.thinking.match(/[\s\S]{1,18}/g).map((t) => ({ type: "thinking_delta", thinking: t })), + { type: "signature_delta", signature: "sig_shot" }, + ); + } + if (turn.text) block(index++, { type: "text", text: "" }, textDeltas(turn.text)); + if (turn.cmd) { + const json = JSON.stringify({ cmd: turn.cmd }); + block( + index++, + { type: "tool_use", id: `toolu_shot_${toolResults + 1}`, name: "exec_command", input: {} }, + (json.match(/[\s\S]{1,32}/g) ?? []).map((partial_json) => ({ + type: "input_json_delta", + partial_json, + })), + ); + finish("tool_use", 160); + } else { + finish("end_turn", 420); + } +} + +function openaiReply(res, body) { + const flat = JSON.stringify(body.messages ?? []); + const script = scriptFor(flat); + const isTitle = flat.includes("concise title"); + const toolResults = flat.split('"role":"tool"').length - 1; + const turn = script.turns[Math.min(toolResults, script.turns.length - 1)]; + + res.writeHead(200, { "content-type": "text/event-stream", "cache-control": "no-cache" }); + const chunk = (delta, finishReason = null, usage) => { + const payload = { + id: "chatcmpl-shot", + object: "chat.completion.chunk", + created: Math.floor(Date.now() / 1000), + model: "deepseek-v4-pro", + choices: [{ index: 0, delta, finish_reason: finishReason }], + }; + if (usage) payload.usage = usage; + res.write(`data: ${JSON.stringify(payload)}\n\n`); + }; + const usage = { + prompt_tokens: 5200, + completion_tokens: turn.cmd ? 180 : 420, + total_tokens: 5620, + prompt_cache_hit_tokens: 4300, + prompt_cache_miss_tokens: 900, + }; + + chunk({ role: "assistant" }); + if (isTitle) { + chunk({ content: script.title }); + chunk({}, "stop", usage); + res.write("data: [DONE]\n\n"); + res.end(); + return; + } + if (turn.thinking) { + for (const t of turn.thinking.match(/[\s\S]{1,18}/g)) chunk({ reasoning_content: t }); + } + if (turn.text) { + for (const t of turn.text.match(/[\s\S]{1,24}/g)) chunk({ content: t }); + } + if (turn.cmd) { + chunk({ + tool_calls: [ + { + index: 0, + id: `call_shot_${toolResults + 1}`, + type: "function", + function: { name: "exec_command", arguments: "" }, + }, + ], + }); + const json = JSON.stringify({ cmd: turn.cmd }); + for (const part of json.match(/[\s\S]{1,32}/g)) { + chunk({ tool_calls: [{ index: 0, function: { arguments: part } }] }); + } + chunk({}, "tool_calls", usage); + } else { + chunk({}, "stop", usage); + } + res.write("data: [DONE]\n\n"); + res.end(); +} + +function startMock() { + const server = http.createServer((req, res) => { + if (req.method !== "POST") { + res.writeHead(404).end(); + return; + } + let body = ""; + req.on("data", (c) => (body += c)); + req.on("end", () => { + let json = {}; + try { + json = JSON.parse(body); + } catch {} + if (req.url?.includes("chat/completions")) return openaiReply(res, json); + if (req.url?.includes("messages")) return anthropicReply(res, json); + console.log(`[mock] unexpected path ${req.url}`); + res.writeHead(404).end(); + }); + }); + return new Promise((resolve) => server.listen(MOCK_PORT, "127.0.0.1", () => resolve(server))); +} + +// --------------------------------------------------------------------------- +// Server + API helpers. +// --------------------------------------------------------------------------- + +async function waitFor(url, tries = 60) { + for (let i = 0; i < tries; i++) { + try { + const res = await fetch(url); + if (res.ok) return; + } catch {} + await new Promise((r) => setTimeout(r, 500)); + } + throw new Error(`server not ready: ${url}`); +} + +async function api(cookie, method, url, body) { + const res = await fetch(`${BASE}${url}`, { + method, + headers: { + "content-type": "application/json", + ...(cookie ? { cookie } : {}), + }, + ...(body ? { body: JSON.stringify(body) } : {}), + }); + if (!res.ok) throw new Error(`${method} ${url} -> ${res.status} ${await res.text()}`); + return { json: await res.json().catch(() => ({})), setCookie: res.headers.get("set-cookie") }; +} + +async function login(userId, password) { + const { json, setCookie } = await api(null, "POST", "/api/auth/login", { userId, password }); + if (!setCookie) throw new Error("no session cookie from login"); + return { cookie: setCookie.split(";")[0], user: json.user }; +} + +/** Per-language demo users so sidebars stay monolingual in the shots. */ +const USERS = { + zh: { + userId: "demo", + agents: [ + { + agentId: "data_analyst", + name: "数据分析师", + description: "面向 CSV / Excel 的数据分析、图表与报表生成", + }, + { agentId: "web_scout", name: "网页调研员", description: "网页检索、信息核对与调研纪要整理" }, + { + agentId: "agent_optimizer", + name: "Agent 优化师", + description: "评估其他 Agent 的表现并迭代其提示词与技能", + }, + ], + }, + en: { + userId: "alex", + agents: [ + { + agentId: "data_analyst", + name: "Data Analyst", + description: "CSV / Excel analysis, charts and report generation", + }, + { + agentId: "web_scout", + name: "Web Scout", + description: "Web research, fact checking and note-taking", + }, + { + agentId: "agent_optimizer", + name: "Agent Optimizer", + description: "Evaluates other Agents and iterates their prompts and Skills", + }, + ], + }, +}; + +/** Provision a user with models + a few Agents; returns { cookie, password, projectId }. */ +async function provisionUser(adminCookie, lang) { + const { userId, agents } = USERS[lang]; + const initial = `${userId}12345`; + await api(adminCookie, "POST", "/api/admin/users", { userId, password: initial }).catch((e) => { + if (!String(e).includes("409")) throw e; + }); + let session = await login(userId, initial); + // Rotate once so the initial-password banner disappears from the shots. + let password = initial; + try { + await api(session.cookie, "PUT", "/api/me/password", { + oldPassword: initial, + newPassword: `penguin-${userId}-2026`, + }); + password = `penguin-${userId}-2026`; + } catch {} + session = await login(userId, password); + + const projects = (await api(session.cookie, "GET", "/api/projects")).json; + const projectId = projects.projects[0].projectId; + + await api(session.cookie, "PUT", `/api/projects/${projectId}/models`, { + defaultModel: { provider: "deepseek", modelId: "deepseek-v4-pro" }, + models: [ + { + provider: "deepseek", + modelId: "deepseek-v4-pro", + apiKey: "sk-demo", + baseUrl: MOCK, + contextWindow: 1000000, + pricing: { cacheRead: 0.003571, cacheWrite: 0.428571, output: 0.857143 }, + }, + ], + }); + + for (const agent of agents) { + await api(session.cookie, "POST", `/api/projects/${projectId}/agents`, agent).catch((e) => { + if (!String(e).includes("409")) throw e; + }); + } + + return { cookie: session.cookie, password, projectId, userId }; +} + +// --------------------------------------------------------------------------- +// Main. +// --------------------------------------------------------------------------- + +const dataRoot = mkdtempSync(path.join(os.tmpdir(), "penguin-shots-")); +const wsDir = path.join(dataRoot, "workspace-apps"); +mkdirSync(wsDir, { recursive: true }); +mkdirSync(OUT_DIR, { recursive: true }); + +const mock = await startMock(); +console.log(`[shots] mock LLM on ${MOCK}`); + +const srv = spawn("node", [path.join(ROOT, "packages/server/dist/index.js")], { + env: { + ...process.env, + PENGUIN_HOME: path.join(dataRoot, "home"), + PENGUIN_WEB_DB: path.join(dataRoot, "web.db"), + PENGUIN_WEB_DIST: path.join(ROOT, "packages/web/dist"), + PORT: String(SRV_PORT), + HOST: "127.0.0.1", + }, + stdio: ["ignore", "pipe", "pipe"], +}); +srv.stderr.on("data", (d) => process.stderr.write(`[srv!] ${d}`)); + +const cleanup = () => { + try { + srv.kill(); + } catch {} + try { + mock.close(); + } catch {} +}; +process.on("exit", cleanup); + +try { + await waitFor(`${BASE}/`); + console.log(`[shots] server ready on ${BASE}`); + + const admin = await login("admin", "admin123"); + const browser = await chromium.launch(); + + // WebP encoder: Chromium re-encodes the PNG screenshot buffer via canvas, which + // keeps repo assets small (~5x lighter than PNG) with no native image deps. + const encoderPage = await browser.newPage(); + async function saveWebp(pngBuffer, fileName) { + const dataUrl = await encoderPage.evaluate(async (b64) => { + const img = new Image(); + img.src = `data:image/png;base64,${b64}`; + await img.decode(); + const canvas = document.createElement("canvas"); + canvas.width = img.width; + canvas.height = img.height; + canvas.getContext("2d").drawImage(img, 0, 0); + return canvas.toDataURL("image/webp", 0.82); + }, pngBuffer.toString("base64")); + writeFileSync(path.join(OUT_DIR, fileName), Buffer.from(dataUrl.split(",")[1], "base64")); + console.log(`[shots] ${fileName}`); + } + + /** The final answer is the only turn mentioning the npm run command. */ + const DONE_MARKER = "npm install && npm start"; + + for (const lang of ["zh", "en"]) { + const user = await provisionUser(admin.cookie, lang); + const script = SCRIPTS[lang]; + + // Language-specific workspace subdir so zh/en runs don't collide on files. + const ws = path.join(wsDir, lang); + mkdirSync(ws, { recursive: true }); + + const sess = ( + await api( + user.cookie, + "POST", + `/api/projects/${user.projectId}/agents/default_agent/sessions`, + { + provider: "deepseek", + modelId: "deepseek-v4-pro", + approvalMode: "allow-all", + workspace: ws, + }, + ) + ).json; + const sessionId = sess.session.sessionId; + + let firstTheme = true; + for (const theme of ["light", "dark"]) { + // 1280x800 @1.5x -> 1920x1200: sharp enough for the landing's ~1024px-wide + // frames on retina, while keeping the WebP assets small. + const context = await browser.newContext({ + viewport: { width: 1280, height: 800 }, + deviceScaleFactor: 1.5, + locale: lang === "zh" ? "zh-CN" : "en-US", + }); + await context.addInitScript( + ([t, l]) => { + localStorage.setItem("penguin.theme", t); + localStorage.setItem("penguin.lang", l); + }, + [theme, lang], + ); + const page = await context.newPage(); + await page.goto(`${BASE}/login`); + const loginRes = await page.request.post(`${BASE}/api/auth/login`, { + data: { userId: user.userId, password: user.password }, + }); + if (!loginRes.ok()) throw new Error(`browser login failed: ${loginRes.status()}`); + + await page.goto(`${BASE}/chat/${sessionId}`); + if (firstTheme) { + // Drive the conversation once per language; the other theme restores it. + const input = page.getByPlaceholder(/输入消息|Type a message/); + await input.waitFor({ timeout: 20000 }); + await input.fill(script.prompt); + await page.getByRole("button", { name: /发送|Send/ }).click(); + firstTheme = false; + } + await page.getByText(DONE_MARKER).first().waitFor({ timeout: 90000 }); + await page.waitForTimeout(2000); + await saveWebp(await page.screenshot(), `chat-${lang}-${theme}.webp`); + + // Trace view: select the session in the list (deep-link selection is unreliable + // right after a fresh navigation, so click explicitly — sidebar shows the same + // title first in DOM order, hence .last()). + await page.goto(`${BASE}/traces?sessionId=${sessionId}`); + await page.waitForTimeout(1500); + await page + .getByText(script.title) + .last() + .click() + .catch(() => {}); + await page.waitForTimeout(2500); + await saveWebp(await page.screenshot(), `traces-${lang}-${theme}.webp`); + + // Evaluation center: open the pre-provisioned example Benchmark scoreboard. + await page.goto(`${BASE}/benchmark`); + await page.waitForTimeout(1500); + await page + .getByText("Example Benchmark") + .first() + .click() + .catch(() => {}); + await page.waitForTimeout(2500); + await saveWebp(await page.screenshot(), `benchmark-${lang}-${theme}.webp`); + + await context.close(); + } + } + + await browser.close(); + console.log(`[shots] done -> ${OUT_DIR}`); + process.exit(0); +} catch (err) { + console.error("[shots] FAILED:", err); + process.exit(1); +} diff --git a/packages/landing/scripts/postbuild.mjs b/packages/landing/scripts/postbuild.mjs new file mode 100644 index 0000000..2ce864a --- /dev/null +++ b/packages/landing/scripts/postbuild.mjs @@ -0,0 +1,13 @@ +/** + * Post-build step for GitHub Pages: copy index.html to 404.html so deep links + * (/blog/xxx) served by Pages' 404 fallback still boot the SPA router, and add + * .nojekyll so Pages serves the dist verbatim without Jekyll processing. + */ +import { copyFileSync, writeFileSync } from "node:fs"; +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; + +const dist = join(dirname(fileURLToPath(import.meta.url)), "..", "dist"); +copyFileSync(join(dist, "index.html"), join(dist, "404.html")); +writeFileSync(join(dist, ".nojekyll"), ""); +console.log("[postbuild] wrote dist/404.html and dist/.nojekyll"); diff --git a/packages/landing/src/app.tsx b/packages/landing/src/app.tsx new file mode 100644 index 0000000..512aee8 --- /dev/null +++ b/packages/landing/src/app.tsx @@ -0,0 +1,20 @@ +/** + * App root: Locale -> Theme -> LocaleScope -> Router provider composition, + * mirroring the Web App (LocaleScope remounts the tree keyed by locale so + * every `S.x` read reflects the active language). + */ +import { LocaleProvider, LocaleScope } from "./state/locale"; +import { ThemeProvider } from "./state/theme"; +import { AppRouter } from "./router"; + +export function App() { + return ( + + + + + + + + ); +} diff --git a/packages/landing/src/assets/shots/benchmark-en-dark.webp b/packages/landing/src/assets/shots/benchmark-en-dark.webp new file mode 100644 index 0000000..d48d05c Binary files /dev/null and b/packages/landing/src/assets/shots/benchmark-en-dark.webp differ diff --git a/packages/landing/src/assets/shots/benchmark-en-light.webp b/packages/landing/src/assets/shots/benchmark-en-light.webp new file mode 100644 index 0000000..05116dd Binary files /dev/null and b/packages/landing/src/assets/shots/benchmark-en-light.webp differ diff --git a/packages/landing/src/assets/shots/benchmark-zh-dark.webp b/packages/landing/src/assets/shots/benchmark-zh-dark.webp new file mode 100644 index 0000000..815428a Binary files /dev/null and b/packages/landing/src/assets/shots/benchmark-zh-dark.webp differ diff --git a/packages/landing/src/assets/shots/benchmark-zh-light.webp b/packages/landing/src/assets/shots/benchmark-zh-light.webp new file mode 100644 index 0000000..70ee6f7 Binary files /dev/null and b/packages/landing/src/assets/shots/benchmark-zh-light.webp differ diff --git a/packages/landing/src/assets/shots/chat-en-dark.webp b/packages/landing/src/assets/shots/chat-en-dark.webp new file mode 100644 index 0000000..5e3567a Binary files /dev/null and b/packages/landing/src/assets/shots/chat-en-dark.webp differ diff --git a/packages/landing/src/assets/shots/chat-en-light.webp b/packages/landing/src/assets/shots/chat-en-light.webp new file mode 100644 index 0000000..6af357b Binary files /dev/null and b/packages/landing/src/assets/shots/chat-en-light.webp differ diff --git a/packages/landing/src/assets/shots/chat-zh-dark.webp b/packages/landing/src/assets/shots/chat-zh-dark.webp new file mode 100644 index 0000000..237e726 Binary files /dev/null and b/packages/landing/src/assets/shots/chat-zh-dark.webp differ diff --git a/packages/landing/src/assets/shots/chat-zh-light.webp b/packages/landing/src/assets/shots/chat-zh-light.webp new file mode 100644 index 0000000..17b4e2a Binary files /dev/null and b/packages/landing/src/assets/shots/chat-zh-light.webp differ diff --git a/packages/landing/src/assets/shots/traces-en-dark.webp b/packages/landing/src/assets/shots/traces-en-dark.webp new file mode 100644 index 0000000..7b50434 Binary files /dev/null and b/packages/landing/src/assets/shots/traces-en-dark.webp differ diff --git a/packages/landing/src/assets/shots/traces-en-light.webp b/packages/landing/src/assets/shots/traces-en-light.webp new file mode 100644 index 0000000..1139619 Binary files /dev/null and b/packages/landing/src/assets/shots/traces-en-light.webp differ diff --git a/packages/landing/src/assets/shots/traces-zh-dark.webp b/packages/landing/src/assets/shots/traces-zh-dark.webp new file mode 100644 index 0000000..ad9cb8c Binary files /dev/null and b/packages/landing/src/assets/shots/traces-zh-dark.webp differ diff --git a/packages/landing/src/assets/shots/traces-zh-light.webp b/packages/landing/src/assets/shots/traces-zh-light.webp new file mode 100644 index 0000000..7a07794 Binary files /dev/null and b/packages/landing/src/assets/shots/traces-zh-light.webp differ diff --git a/packages/landing/src/components/browser-frame.tsx b/packages/landing/src/components/browser-frame.tsx new file mode 100644 index 0000000..a752f70 --- /dev/null +++ b/packages/landing/src/components/browser-frame.tsx @@ -0,0 +1,31 @@ +/** Fake browser chrome around product screenshots: traffic dots + address pill. */ +import type { ReactNode } from "react"; + +export function BrowserFrame({ + children, + url = "127.0.0.1:7364", + className = "", +}: { + children: ReactNode; + url?: string; + className?: string; +}) { + return ( +
+
+
+ {children} +
+ ); +} diff --git a/packages/landing/src/components/category-badge.tsx b/packages/landing/src/components/category-badge.tsx new file mode 100644 index 0000000..7a8c0c4 --- /dev/null +++ b/packages/landing/src/components/category-badge.tsx @@ -0,0 +1,18 @@ +/** Blog category badge: product news (brand tint) vs release notes (neutral tint). */ +import { S } from "../lib/strings"; +import type { BlogCategory } from "../lib/blog"; + +export function CategoryBadge({ category }: { category: BlogCategory }) { + const isNews = category === "news"; + return ( + + {isNews ? S.blog.news : S.blog.changelog} + + ); +} diff --git a/packages/landing/src/components/code-card.tsx b/packages/landing/src/components/code-card.tsx new file mode 100644 index 0000000..3d572af --- /dev/null +++ b/packages/landing/src/components/code-card.tsx @@ -0,0 +1,51 @@ +/** + * Terminal-style code card: rounded border, a slim header with a label + copy button, + * and a monospace body. Comment lines (starting with #) render muted; no highlighter + * dependency — landing snippets are short shell commands. + */ +import { TerminalIcon } from "./icons"; +import { CopyButton } from "./copy-button"; + +export function CodeCard({ + code, + label, + className = "", +}: { + code: string; + label?: string; + className?: string; +}) { + const lines = code.split("\n"); + return ( +
+
+ + + {label ?? "shell"} + + +
+
+        
+          {lines.map((line, i) => {
+            const isComment = line.trimStart().startsWith("#");
+            return (
+              
+                {line.length > 0 ? line : " "}
+              
+            );
+          })}
+        
+      
+
+ ); +} diff --git a/packages/landing/src/components/copy-button.tsx b/packages/landing/src/components/copy-button.tsx new file mode 100644 index 0000000..0819f44 --- /dev/null +++ b/packages/landing/src/components/copy-button.tsx @@ -0,0 +1,50 @@ +/** Copy-to-clipboard button with a transient "copied" state. */ +import { useEffect, useRef, useState } from "react"; +import { S } from "../lib/strings"; +import { CheckIcon, CopyIcon } from "./icons"; + +export function CopyButton({ text, className = "" }: { text: string; className?: string }) { + const [copied, setCopied] = useState(false); + const timer = useRef | null>(null); + + useEffect( + () => () => { + if (timer.current) clearTimeout(timer.current); + }, + [], + ); + + const onCopy = async () => { + try { + await navigator.clipboard.writeText(text); + } catch { + // Clipboard API unavailable (e.g. non-secure context): fall back to a hidden textarea. + const ta = document.createElement("textarea"); + ta.value = text; + document.body.appendChild(ta); + ta.select(); + document.execCommand("copy"); + ta.remove(); + } + setCopied(true); + if (timer.current) clearTimeout(timer.current); + timer.current = setTimeout(() => setCopied(false), 1600); + }; + + return ( + + ); +} diff --git a/packages/landing/src/components/footer.tsx b/packages/landing/src/components/footer.tsx new file mode 100644 index 0000000..873d427 --- /dev/null +++ b/packages/landing/src/components/footer.tsx @@ -0,0 +1,93 @@ +/** Site footer: brand + product/resource link columns + copyright. */ +import { Link, useLocation } from "react-router"; +import { S } from "../lib/strings"; +import { DOCS_URL, LICENSE_URL, RELEASES_URL, REPO_URL } from "../lib/links"; + +export function Footer() { + const { pathname } = useLocation(); + const onHome = pathname === "/"; + const anchor = (id: string, label: string) => + onHome ? ( + + {label} + + ) : ( + + {label} + + ); + + return ( + + ); +} diff --git a/packages/landing/src/components/harness-logo.tsx b/packages/landing/src/components/harness-logo.tsx new file mode 100644 index 0000000..3fbcfc8 --- /dev/null +++ b/packages/landing/src/components/harness-logo.tsx @@ -0,0 +1,30 @@ +/** + * Harness brand marks for the benchmark comparison: PenguinHarness uses the product + * logo asset; Claude Code and Codex use their vendors' marks (simple-icons path data, + * rendered in currentColor so they stay neutral in both themes). + */ + +export type HarnessKind = "penguin" | "claude" | "codex"; + +const ANTHROPIC_PATH = + "M17.3041 3.541h-3.6718l6.696 16.918H24Zm-10.6082 0L0 20.459h3.7442l1.3693-3.5527h7.0052l1.3693 3.5527h3.7442L10.5359 3.541Zm-.3712 10.2232 2.2914-5.9456 2.2914 5.9456Z"; + +const OPENAI_PATH = + "M22.2819 9.8211a5.9847 5.9847 0 0 0-.5157-4.9108 6.0462 6.0462 0 0 0-6.5098-2.9A6.0651 6.0651 0 0 0 4.9807 4.1818a5.9847 5.9847 0 0 0-3.9977 2.9 6.0462 6.0462 0 0 0 .7427 7.0966 5.98 5.98 0 0 0 .511 4.9107 6.051 6.051 0 0 0 6.5146 2.9001A5.9847 5.9847 0 0 0 13.2599 24a6.0557 6.0557 0 0 0 5.7718-4.2058 5.9894 5.9894 0 0 0 3.9977-2.9001 6.0557 6.0557 0 0 0-.7475-7.073zM13.2599 22.4301a4.4755 4.4755 0 0 1-2.8764-1.04l.1419-.0804 4.7783-2.7582a.7948.7948 0 0 0 .3927-.6813v-6.7369l2.02 1.1686a.071.071 0 0 1 .038.052v5.5826a4.504 4.504 0 0 1-4.4945 4.4944zm-9.6607-4.1254a4.4708 4.4708 0 0 1-.5346-3.0137l.142.0852 4.783 2.7582a.7712.7712 0 0 0 .7806 0l5.8428-3.3685v2.3324a.0804.0804 0 0 1-.0332.0615L9.74 19.9502a4.4992 4.4992 0 0 1-6.1408-1.6464zM2.3408 7.8956a4.485 4.485 0 0 1 2.3655-1.9728V11.6a.7664.7664 0 0 0 .3879.6765l5.8144 3.3543-2.0201 1.1685a.0757.0757 0 0 1-.071 0l-4.8303-2.7865A4.504 4.504 0 0 1 2.3408 7.8956zm16.5963 3.8558L13.1038 8.364 15.1192 7.2a.0757.0757 0 0 1 .071 0l4.8303 2.7913a4.4944 4.4944 0 0 1-.6765 8.1042v-5.6772a.79.79 0 0 0-.407-.667zm2.0107-3.0231-.142-.0852-4.7735-2.7818a.7759.7759 0 0 0-.7854 0L9.409 9.2297V6.8974a.0662.0662 0 0 1 .0284-.0615l4.8303-2.7866a4.4992 4.4992 0 0 1 6.6802 4.66zM8.3065 12.863l-2.02-1.1638a.0804.0804 0 0 1-.038-.0567V6.0742a4.4992 4.4992 0 0 1 7.3757-3.4537l-.142.0805L8.704 5.459a.7948.7948 0 0 0-.3927.6813zm1.0976-2.3654 2.602-1.4998 2.6069 1.4998v2.9994l-2.5974 1.4997-2.6067-1.4997Z"; + +export function HarnessLogo({ + kind, + className = "h-3.5 w-3.5", +}: { + kind: HarnessKind; + className?: string; +}) { + if (kind === "penguin") { + return ; + } + return ( + + ); +} diff --git a/packages/landing/src/components/icons.tsx b/packages/landing/src/components/icons.tsx new file mode 100644 index 0000000..5da366f --- /dev/null +++ b/packages/landing/src/components/icons.tsx @@ -0,0 +1,332 @@ +/** + * Inline icon set (lucide-style 24x24 stroke icons + the GitHub mark). Kept local so + * the landing page has zero icon dependencies; all icons inherit currentColor. + */ +import type { ReactNode, SVGProps } from "react"; + +type IconProps = SVGProps; + +function Icon({ children, ...props }: IconProps & { children: ReactNode }) { + return ( + + ); +} + +export function GitHubIcon(props: IconProps) { + return ( + + ); +} + +export function SunIcon(props: IconProps) { + return ( + + + + + ); +} + +export function MoonIcon(props: IconProps) { + return ( + + + + ); +} + +export function MonitorIcon(props: IconProps) { + return ( + + + + + ); +} + +export function CopyIcon(props: IconProps) { + return ( + + + + + ); +} + +export function CheckIcon(props: IconProps) { + return ( + + + + ); +} + +export function MenuIcon(props: IconProps) { + return ( + + + + ); +} + +export function XIcon(props: IconProps) { + return ( + + + + ); +} + +export function ArrowRightIcon(props: IconProps) { + return ( + + + + ); +} + +export function TerminalIcon(props: IconProps) { + return ( + + + + ); +} + +export function DownloadIcon(props: IconProps) { + return ( + + + + ); +} + +export function SlidersIcon(props: IconProps) { + return ( + + + + ); +} + +export function PlayIcon(props: IconProps) { + return ( + + + + ); +} + +export function FeatherIcon(props: IconProps) { + return ( + + + + ); +} + +export function BotIcon(props: IconProps) { + return ( + + + + + + ); +} + +export function RefreshIcon(props: IconProps) { + return ( + + + + + ); +} + +export function CpuIcon(props: IconProps) { + return ( + + + + + + ); +} + +export function LayersIcon(props: IconProps) { + return ( + + + + ); +} + +export function BlocksIcon(props: IconProps) { + return ( + + + + + + ); +} + +export function ShareIcon(props: IconProps) { + return ( + + + + + + + ); +} + +export function SparklesIcon(props: IconProps) { + return ( + + + + + ); +} + +export function ActivityIcon(props: IconProps) { + return ( + + + + ); +} + +export function BarChartIcon(props: IconProps) { + return ( + + + + ); +} + +export function HistoryIcon(props: IconProps) { + return ( + + + + ); +} + +export function ClockIcon(props: IconProps) { + return ( + + + + + ); +} + +export function PieChartIcon(props: IconProps) { + return ( + + + + ); +} + +export function ShieldIcon(props: IconProps) { + return ( + + + + ); +} + +export function ShieldCheckIcon(props: IconProps) { + return ( + + + + + ); +} + +export function FrameIcon(props: IconProps) { + return ( + + + + ); +} + +export function FileCheckIcon(props: IconProps) { + return ( + + + + + ); +} + +export function MessageSquareIcon(props: IconProps) { + return ( + + + + ); +} + +export function UsersIcon(props: IconProps) { + return ( + + + + + + ); +} + +export function GlobeIcon(props: IconProps) { + return ( + + + + + ); +} + +export function ChevronDownIcon(props: IconProps) { + return ( + + + + ); +} + +export function ExternalLinkIcon(props: IconProps) { + return ( + + + + ); +} + +export function KeyIcon(props: IconProps) { + return ( + + + + ); +} diff --git a/packages/landing/src/components/lang-toggle.tsx b/packages/landing/src/components/lang-toggle.tsx new file mode 100644 index 0000000..28194fe --- /dev/null +++ b/packages/landing/src/components/lang-toggle.tsx @@ -0,0 +1,74 @@ +/** + * Language menu: 中文 / English / follow system, persisted via the locale context. + * A small dropdown (globe + current label); closes on outside click or selection. + * Scroll position across the locale remount is preserved by LocaleScope. + */ +import { useEffect, useRef, useState } from "react"; +import { useLocale } from "../state/locale"; +import type { LangPref } from "../state/locale"; +import { S } from "../lib/strings"; +import { CheckIcon, ChevronDownIcon, GlobeIcon } from "./icons"; + +export function LangToggle() { + const { lang, setLang } = useLocale(); + const [open, setOpen] = useState(false); + const ref = useRef(null); + + useEffect(() => { + if (!open) return; + const onDown = (e: MouseEvent) => { + if (ref.current && !ref.current.contains(e.target as Node)) setOpen(false); + }; + document.addEventListener("mousedown", onDown); + return () => document.removeEventListener("mousedown", onDown); + }, [open]); + + const OPTIONS: Array<{ value: LangPref; label: string }> = [ + { value: "en", label: S.lang.en }, + { value: "zh", label: S.lang.zh }, + { value: "system", label: S.lang.system }, + ]; + const current = OPTIONS.find((o) => o.value === lang) ?? OPTIONS[2]!; + + return ( +
+ + {open && ( +
+ {OPTIONS.map((o) => ( + + ))} +
+ )} +
+ ); +} diff --git a/packages/landing/src/components/nav.tsx b/packages/landing/src/components/nav.tsx new file mode 100644 index 0000000..636d3f2 --- /dev/null +++ b/packages/landing/src/components/nav.tsx @@ -0,0 +1,154 @@ +/** + * Sticky top navigation: logo + section anchors + blog link + language/theme toggles + + * GitHub. Desktop links share a sliding hover pill (position animated between items); + * section anchors use plain hashes on the home page (native smooth scroll) and route + * back to "/#id" from other pages; a disclosure menu covers small screens. + */ +import { useState } from "react"; +import type { MouseEvent } from "react"; +import { Link, useLocation } from "react-router"; +import { S } from "../lib/strings"; +import { DOCS_URL, REPO_URL } from "../lib/links"; +import { GitHubIcon, MenuIcon, XIcon } from "./icons"; +import { ThemeToggle } from "./theme-toggle"; +import { LangToggle } from "./lang-toggle"; + +const SECTION_IDS = ["highlights", "quickstart", "benchmark", "contract", "features"] as const; + +interface Indicator { + left: number; + width: number; +} + +export function Nav() { + const { pathname } = useLocation(); + const onHome = pathname === "/"; + const [open, setOpen] = useState(false); + const [indicator, setIndicator] = useState(null); + + const sectionLabel: Record<(typeof SECTION_IDS)[number], string> = { + highlights: S.nav.highlights, + quickstart: S.nav.quickstart, + benchmark: S.nav.benchmark, + contract: S.nav.contract, + features: S.nav.features, + }; + + // Mobile menu links keep their own hover backgrounds; desktop links rely on the pill. + const mobileLinkCls = + "rounded-md px-2.5 py-1.5 text-sm text-gray-600 transition-colors hover:bg-gray-50 hover:text-gray-900 dark:text-gray-400 dark:hover:bg-gray-900 dark:hover:text-gray-100"; + const deskLinkCls = + "relative z-10 rounded-md px-2.5 py-1.5 text-sm text-gray-600 transition-colors hover:text-gray-900 dark:text-gray-400 dark:hover:text-gray-100"; + + const slideTo = (e: MouseEvent) => { + const el = e.currentTarget; + setIndicator({ left: el.offsetLeft, width: el.offsetWidth }); + }; + + const desktopLinks = ( + <> + {SECTION_IDS.map((id) => + onHome ? ( + + {sectionLabel[id]} + + ) : ( + + {sectionLabel[id]} + + ), + )} + + {S.nav.blog} + + {/* Docs is a sibling SPA under /docs/ — a plain anchor, not a router Link. */} + + {S.nav.docs} + + + ); + + const mobileLinks = ( + <> + {SECTION_IDS.map((id) => + onHome ? ( + setOpen(false)}> + {sectionLabel[id]} + + ) : ( + setOpen(false)}> + {sectionLabel[id]} + + ), + )} + setOpen(false)}> + {S.nav.blog} + + setOpen(false)}> + {S.nav.docs} + + + ); + + return ( +
+
+ setOpen(false)}> + + {S.siteName} + + + + +
+ + + + + + +
+
+ + {open && ( + + )} +
+ ); +} diff --git a/packages/landing/src/components/neon-bg.tsx b/packages/landing/src/components/neon-bg.tsx new file mode 100644 index 0000000..0eb2564 --- /dev/null +++ b/packages/landing/src/components/neon-bg.tsx @@ -0,0 +1,15 @@ +/** + * Ambient neon backdrop: faint blurred glows drifting slowly behind the whole page, + * anchored to the document (the layout root is position:relative) so they scroll away + * with the content. Styles in styles.css; reduced-motion freezes them. Decorative only. + */ +export function NeonBackground() { + return ( + + ); +} diff --git a/packages/landing/src/components/section.tsx b/packages/landing/src/components/section.tsx new file mode 100644 index 0000000..4e32172 --- /dev/null +++ b/packages/landing/src/components/section.tsx @@ -0,0 +1,47 @@ +/** Shared section shell: anchor id + centered header (eyebrow / title / subtitle) + content. */ +import type { ReactNode } from "react"; +import { useReveal } from "../lib/reveal"; + +export function Section({ + id, + eyebrow, + title, + subtitle, + children, + className = "", +}: { + id?: string; + eyebrow?: string; + title?: string; + subtitle?: string; + children: ReactNode; + className?: string; +}) { + const ref = useReveal(); + return ( +
+
+ {(eyebrow || title || subtitle) && ( +
+ {eyebrow && ( +

+ {eyebrow} +

+ )} + {title && ( +

+ {title} +

+ )} + {subtitle && ( +

+ {subtitle} +

+ )} +
+ )} + {children} +
+
+ ); +} diff --git a/packages/landing/src/components/theme-toggle.tsx b/packages/landing/src/components/theme-toggle.tsx new file mode 100644 index 0000000..4d552ee --- /dev/null +++ b/packages/landing/src/components/theme-toggle.tsx @@ -0,0 +1,24 @@ +/** Theme cycle button: light -> dark -> system, icon reflects the current mode. */ +import { useTheme } from "../state/theme"; +import type { ThemeMode } from "../state/theme"; +import { S } from "../lib/strings"; +import { MonitorIcon, MoonIcon, SunIcon } from "./icons"; + +const NEXT: Record = { light: "dark", dark: "system", system: "light" }; + +export function ThemeToggle() { + const { mode, setMode } = useTheme(); + const label = mode === "light" ? S.theme.light : mode === "dark" ? S.theme.dark : S.theme.system; + const IconCmp = mode === "light" ? SunIcon : mode === "dark" ? MoonIcon : MonitorIcon; + return ( + + ); +} diff --git a/packages/landing/src/lib/benchmark-data.ts b/packages/landing/src/lib/benchmark-data.ts new file mode 100644 index 0000000..535553c --- /dev/null +++ b/packages/landing/src/lib/benchmark-data.ts @@ -0,0 +1,99 @@ +/** + * Benchmark data, two suites driven by the same DeepSeek V4 Pro model, published in + * one unified shape: framework / model / accuracy (%) / Tokens (M) / cost ($), all + * as per-run means. Suite specifics (case count, runs, thinking level, timeout, + * pricing source) live in the footnote strings, not in the table. + * - Data analysis: 15 tasks, single run, USD pricing. + * - Coding: 40 tasks x 2 runs averaged; official CNY pricing converted at $1 = ¥7 + * (0.289 / 0.338 / 0.299 CNY per run). + */ +import type { HarnessKind } from "../components/harness-logo"; + +export interface BenchResult { + kind: HarnessKind; + framework: string; + model: string; + accuracyPct: number; + tokensM: number; + costUsd: number; + /** The series the story is about (emphasis form: accent hue vs de-emphasis gray). */ + emphasized?: boolean; +} + +const MODEL = "DeepSeek V4 Pro"; + +export const DATA_BENCH: BenchResult[] = [ + { + kind: "penguin", + framework: "PenguinHarness", + model: MODEL, + accuracyPct: 66.7, + tokensM: 18.037757, + costUsd: 0.552406, + emphasized: true, + }, + { + kind: "claude", + framework: "Claude Code", + model: MODEL, + accuracyPct: 66.7, + tokensM: 21.166305, + costUsd: 0.640706, + }, + { + kind: "codex", + framework: "OpenAI Codex", + model: MODEL, + accuracyPct: 46.7, + tokensM: 13.362259, + costUsd: 0.427011, + }, +]; + +export const CODE_BENCH: BenchResult[] = [ + { + kind: "penguin", + framework: "PenguinHarness", + model: MODEL, + accuracyPct: 50.0, + tokensM: 2.1, + costUsd: 0.0413, + emphasized: true, + }, + { + kind: "claude", + framework: "Claude Code", + model: MODEL, + accuracyPct: 48.75, + tokensM: 2.0, + costUsd: 0.0483, + }, + { + kind: "codex", + framework: "OpenAI Codex", + model: MODEL, + accuracyPct: 42.5, + tokensM: 2.65, + costUsd: 0.0427, + }, +]; + +/** 66.7 -> "66.7%" (chart caps). */ +export function formatPct(pct: number): string { + return `${pct.toFixed(1)}%`; +} + +/** Table accuracy at suite precision: 66.7 (1dp) vs 48.75 (2dp). */ +export function formatAccuracy(pct: number, dp: number): string { + return pct.toFixed(dp); +} + +/** 18.037757 -> "18.0M" / 2.65 -> "2.65M" (chart caps at suite precision). */ +export function formatTokensM(tokens: number, dp = 1): string { + return `${tokens.toFixed(dp)}M`; +} + +/** Chart cost caps at suite precision: $0.55 vs $0.041. */ +export function formatUsd(cost: number, dp = 2): string { + return `$${cost.toFixed(dp)}`; +} diff --git a/packages/landing/src/lib/blog.ts b/packages/landing/src/lib/blog.ts new file mode 100644 index 0000000..a2fd7d1 --- /dev/null +++ b/packages/landing/src/lib/blog.ts @@ -0,0 +1,67 @@ +/** + * Blog index: local Markdown posts imported at build time via import.meta.glob. + * File naming: content/blog/..md — one file per post per language; + * a post missing the active language falls back to the other one, so the list + * is always complete in both locales. + */ +import { parseFrontmatter } from "./frontmatter"; +import type { Locale } from "../state/locale"; + +export type BlogCategory = "news" | "changelog"; + +export interface BlogPost { + slug: string; + lang: Locale; + title: string; + /** YYYY-MM-DD */ + date: string; + category: BlogCategory; + excerpt: string; + body: string; +} + +const files = import.meta.glob("../../content/blog/*.md", { + query: "?raw", + import: "default", + eager: true, +}) as Record; + +function toPost(path: string, raw: string): BlogPost | null { + const file = path.split("/").pop() ?? ""; + const match = /^(.+)\.(zh|en)\.md$/.exec(file); + if (!match) return null; + const { meta, body } = parseFrontmatter(raw); + const category: BlogCategory = meta.category === "changelog" ? "changelog" : "news"; + return { + slug: match[1]!, + lang: match[2] as Locale, + title: meta.title ?? match[1]!, + date: meta.date ?? "", + category, + excerpt: meta.excerpt ?? "", + body, + }; +} + +const ALL: BlogPost[] = Object.entries(files) + .map(([path, raw]) => toPost(path, raw)) + .filter((p): p is BlogPost => p !== null); + +/** Pick the locale's version of each slug (fallback to the other language), newest first. */ +export function postsFor(locale: Locale, category?: BlogCategory): BlogPost[] { + const bySlug = new Map(); + for (const post of ALL) { + const existing = bySlug.get(post.slug); + if (!existing || (existing.lang !== locale && post.lang === locale)) { + bySlug.set(post.slug, post); + } + } + return [...bySlug.values()] + .filter((p) => (category ? p.category === category : true)) + .sort((a, b) => (a.date < b.date ? 1 : a.date > b.date ? -1 : a.slug.localeCompare(b.slug))); +} + +export function getPost(slug: string, locale: Locale): BlogPost | undefined { + const candidates = ALL.filter((p) => p.slug === slug); + return candidates.find((p) => p.lang === locale) ?? candidates[0]; +} diff --git a/packages/landing/src/lib/frontmatter.ts b/packages/landing/src/lib/frontmatter.ts new file mode 100644 index 0000000..8d2b212 --- /dev/null +++ b/packages/landing/src/lib/frontmatter.ts @@ -0,0 +1,31 @@ +/** + * Minimal frontmatter parser for blog posts: a leading `---` block of `key: value` + * lines (values may contain colons; quotes optional). Kept dependency-free and pure + * so it is unit-testable without Vite. + */ + +export interface Frontmatter { + meta: Record; + body: string; +} + +export function parseFrontmatter(raw: string): Frontmatter { + const normalized = raw.replace(/\r\n/g, "\n"); + const match = /^---\n([\s\S]*?)\n---\n?/.exec(normalized); + if (!match) return { meta: {}, body: normalized.trim() }; + const meta: Record = {}; + for (const line of match[1]!.split("\n")) { + const idx = line.indexOf(":"); + if (idx === -1) continue; + const key = line.slice(0, idx).trim(); + let value = line.slice(idx + 1).trim(); + if ( + (value.startsWith('"') && value.endsWith('"')) || + (value.startsWith("'") && value.endsWith("'")) + ) { + value = value.slice(1, -1); + } + if (key) meta[key] = value; + } + return { meta, body: normalized.slice(match[0].length).trim() }; +} diff --git a/packages/landing/src/lib/links.ts b/packages/landing/src/lib/links.ts new file mode 100644 index 0000000..384c94d --- /dev/null +++ b/packages/landing/src/lib/links.ts @@ -0,0 +1,19 @@ +/** External links and language-independent constants used across the landing page. */ + +export const REPO_URL = "https://github.com/Prism-Shadow/penguin-harness"; +export const RELEASES_URL = `${REPO_URL}/releases`; +export const LICENSE_URL = `${REPO_URL}/blob/main/LICENSE`; + +/** + * Docs site: a sibling SPA deployed under the landing page's own base ("//docs/", + * see scripts/build-site.mjs). A plain href — it is a separate app, not a router route. + * In local dev this resolves to "/docs/", which only exists in the assembled build. + */ +export const DOCS_URL = `${import.meta.env.BASE_URL}docs/`; + +/** One-line installer (Linux / macOS, x64 / arm64, bundled Node runtime). */ +export const INSTALL_CMD = `curl -fsSL ${REPO_URL}/releases/latest/download/install.sh | sh`; + +/** API key consoles (same URLs the in-app Models page links to). */ +export const DEEPSEEK_KEYS_URL = "https://platform.deepseek.com/api_keys"; +export const OPENROUTER_KEYS_URL = "https://openrouter.ai/workspaces/default/keys"; diff --git a/packages/landing/src/lib/reveal.ts b/packages/landing/src/lib/reveal.ts new file mode 100644 index 0000000..471960d --- /dev/null +++ b/packages/landing/src/lib/reveal.ts @@ -0,0 +1,34 @@ +/** + * Scroll reveal: adds .reveal on mount and .reveal-visible once the element enters + * the viewport (one-shot). Reduced-motion users see content immediately via the CSS + * override in styles.css. + */ +import { useEffect, useRef } from "react"; +import type { RefObject } from "react"; + +export function useReveal(): RefObject { + const ref = useRef(null); + useEffect(() => { + const el = ref.current; + if (!el) return; + el.classList.add("reveal"); + if (!("IntersectionObserver" in window)) { + el.classList.add("reveal-visible"); + return; + } + const io = new IntersectionObserver( + (entries) => { + for (const e of entries) { + if (e.isIntersecting) { + el.classList.add("reveal-visible"); + io.disconnect(); + } + } + }, + { rootMargin: "0px 0px -10% 0px" }, + ); + io.observe(el); + return () => io.disconnect(); + }, []); + return ref; +} diff --git a/packages/landing/src/lib/strings-en.ts b/packages/landing/src/lib/strings-en.ts new file mode 100644 index 0000000..99e57fd --- /dev/null +++ b/packages/landing/src/lib/strings-en.ts @@ -0,0 +1,353 @@ +/** + * English dictionary (constrained by the `Strings` type to the same shape as zh): + * locale switching goes through state/locale.tsx. Keep domain term capitalization + * consistent with zh — Agent, Workspace, Token, Task, Skill, Trace, etc. + */ +import type { Strings } from "./strings"; + +export const en: Strings = { + siteName: "PenguinHarness", + + nav: { + highlights: "Highlights", + quickstart: "Quick start", + benchmark: "Benchmark", + contract: "CONTRACT.md", + features: "Features", + blog: "Blog", + docs: "Docs", + github: "GitHub", + openMenu: "Open menu", + closeMenu: "Close menu", + }, + + theme: { + label: "Theme", + light: "Light", + dark: "Dark", + system: "System", + }, + + lang: { + label: "Language", + zh: "中文", + en: "English", + system: "System", + }, + + hero: { + badge: "Build AI agents, with an Agent", + titlePrefix: "Efficient Self-Improving Harness for ", + titleWords: ["Developers", "Enterprises"], + titleSuffix: "", + titleSuffixNoWrap: "", + keywords: ["Lightweight", "Efficient", "Open Source"], + ctaPrimary: "Get started", + ctaGithub: "GitHub", + installHint: + "One-line install (Linux / macOS, x64 / arm64, bundled Node runtime — unpack and run)", + stats: [ + { value: "1000+", label: "supported models" }, + { value: "1×CPU", label: "minimum footprint" }, + { value: "100%", label: "open source, local deploy" }, + { value: "First native", label: "recursively self-improving harness" }, + ], + }, + + copy: { + copy: "Copy", + copied: "Copied", + }, + + pillars: { + eyebrow: "Three pillars", + title: "Built for building — and evolving — agents", + subtitle: + "PenguinHarness is the first open-source harness to ship “agents building agents” and recursive self-improvement.", + root: "PenguinHarness", + concepts: ["Penguin Message", "Penguin SDK", "Penguin Skills"], + diagramLabel: + "PenguinHarness radiates into Penguin Message, Penguin SDK and Penguin Skills, each extending into one pillar", + items: [ + { + title: "Simplest Is the Best", + tag: "", + desc: "A deliberately minimal toolset over clean low-level interfaces: fewer tool calls, fewer Tokens, complex tasks done efficiently.", + }, + { + title: "Harness for Building Agents", + tag: "", + desc: "With the PenguinHarness SDK, an Agent builds complete Agent applications for you — autonomously, from scratch.", + }, + { + title: "Harness for Recursive Self-Improvement", + tag: "", + desc: "With PenguinHarness Skills, an Agent evaluates and optimizes itself, improving recursively over time.", + }, + ], + }, + + selfImprove: { + eyebrow: "The self-improvement loop", + title: "Multi-agent collaboration makes evolution automatic", + subtitle: + "The Optimizer orchestrates multiple Evaluators to score the Target Agent in parallel, uses the scores and run traces to find where points were lost, and upgrades the Agent from version N to N+1 — with a snapshot before every round.", + nodeOptimizer: "Optimizer", + nodeEvaluator: "Evaluator × N", + nodeTarget: "Target Agent", + badgeOld: "vN", + badgeNew: "vN+1", + edgeSpawn: "spawn parallel evaluations", + edgeBench: "run Benchmarks", + edgeFeedback: "scores & traces", + edgeImprove: "update prompts & Skills", + trends: [ + { label: "Score", hint: "keeps rising" }, + { label: "Cost", hint: "keeps falling" }, + { label: "Time", hint: "keeps shrinking" }, + ], + diagramLabel: + "Self-improvement loop: the Optimizer orchestrates Evaluators to score, then upgrades the Target Agent from vN to vN+1 via scores and traces", + }, + + quickstart: { + eyebrow: "Quick start", + title: "Your first task in three steps", + subtitle: + "Install with one command and let the Agent work from a desktop-grade interface — all data stays in your local ~/.penguin/data directory.", + step1: "Install", + step1Desc: + "Linux / macOS (x64 / arm64) with a bundled Node runtime — unpack and run; upgrades never touch your data.", + tabWeb: "Web UI", + tabCli: "CLI", + webStep2: "Open the web interface", + webStep2Desc: + "penguin web starts the local service and opens your browser; sign in with the built-in admin account admin / admin123 (change the password right after).", + webCmd: "penguin web # opens http://127.0.0.1:7364", + webStep3: "Configure a model in the UI and start chatting", + webStep3Desc: + "Open the Models page, paste an API key under the DeepSeek or OpenRouter group and set it as default; then head back to Chat and hand the Agent its first task — e.g. “Analyze data.csv and summarize quarterly sales”.", + getKeyPrefix: "Get an API key: ", + getDeepseekKey: "DeepSeek console", + getOpenrouterKey: "OpenRouter console", + cliStep2: "Configure a model", + cliStep2Desc: + "Using the DeepSeek official API or the OpenRouter gateway as examples — one command configures it and sets the default.", + tabDeepseek: "DeepSeek", + tabOpenrouter: "OpenRouter", + deepseekCmd: `penguin config model add \\ + --model-id deepseek-v4-pro \\ + --api-key sk-your-deepseek-key \\ + --set-default`, + deepseekNote: + "The provider is inferred as deepseek automatically; omit --api-key to fall back to the DEEPSEEK_API_KEY environment variable.", + openrouterCmd: `penguin config model add \\ + --provider openrouter \\ + --model-id deepseek/deepseek-v4-pro \\ + --api-key sk-or-your-key \\ + --set-default`, + openrouterNote: + "Gateway groups pre-fill the OpenAI-compatible protocol and base URL — one key unlocks a thousand models.", + cliStep3: "Run", + cliStep3Desc: + "penguin run executes a single task; penguin chat drops you into an interactive REPL.", + runCmd: `penguin run --approve allow-all \\ + --message "Analyze data.csv and summarize quarterly sales"`, + }, + + showcase: { + eyebrow: "Use cases", + title: "Daily tasks + zero-code AI development", + subtitle: + "Hand recurring chores to an Agent that runs around the clock — and go from a one-sentence request to a runnable Agent app without writing a line of code.", + tagChat: "Zero-code AI development", + captionChat: + "Development by conversation: the Agent builds an Agent app from scratch, tools really execute", + tagTraces: "Daily tasks", + captionTraces: + "Autonomous runs, fully replayable: every request and tool call with usage and timing", + tagBenchmark: "Continuous evolution", + captionBenchmark: "Built-in Benchmark scoreboards — scores climb with every round", + }, + + contract: { + eyebrow: "A contract for stable evolution", + title: "CONTRACT.md", + subtitle: + "PenguinHarness treats this contract as the boundary and bedrock of evolution: capability may grow, the boundary never drifts.", + intro: + "Evolution needs boundaries. The contract is the covenant between harness and Agent: capability grows within; the boundary holds without.", + items: [ + { + term: "Working boundary", + text: "Every Agent runs on the same harness: Sessions are created under an Agent, Tasks run inside a Session; self-improvement happens only inside Workspace and Skills, while the harness kernel and its safety mechanisms never change.", + }, + { + term: "Editable files", + text: "An Agent's prompts, Skills and configuration live as editable files on disk, never as constants baked into code. What you can see, the Agent can improve; what you can edit, it can learn.", + }, + { + term: "Full tracing", + text: "Every model request and every tool call is written to the Trace in full: how many Tokens it spent, how long it took, why it failed — all replayable line by line afterwards.", + }, + { + term: "Approvals & audit", + text: "Every tool call passes approval before it runs, and every decision leaves an audit record — what the Agent did is never a mystery.", + }, + { + term: "Version control", + text: "Before each optimization, the Agent State is snapshotted. If a round fails or regresses, restore any historical version in one step.", + }, + { + term: "Progressive loading", + text: "Content for the model is indexed first and read on demand — never dumped wholesale into context. The cleaner the context, the steadier the behavior.", + }, + { + term: "Error handling", + text: "Errors split into retryable and fatal: retryable ones retry automatically, fatal ones converge into messages the model can see and react to. No task dies of a single failure.", + }, + { + term: "Credential isolation", + text: "API keys and other credentials live in hidden files and move only through system interfaces — never entering model context, never shown in plain text.", + }, + { + term: "Model decoupling", + text: "Models are not bound to Agents: switch to a stronger or cheaper model at any time without rewriting the Agent.", + }, + { + term: "Recoverable trajectories", + text: "Any Session can be fully restored from its Trace: restart the process or move machines without losing context.", + }, + ], + outro: "The contract does not cap what an Agent can become — only how it gets there.", + }, + + benchmark: { + eyebrow: "Benchmark", + title: "Same model, equal or better quality, lower cost", + subtitle: + "All runs use the same DeepSeek V4 Pro model — head-to-head against Claude Code and OpenAI Codex on two suites.", + higherBetter: "higher is better", + lowerBetter: "lower is better", + dimScore: "Accuracy", + dimTokens: "Tokens", + dimCost: "Cost", + dataTitle: "Complex data analysis", + dataDesc: + "Ties Claude Code on accuracy and clearly beats OpenAI Codex — with fewer Tokens at lower cost.", + dataFootnote: + "15 complex data-analysis tasks · averaged over 1 run · cost estimated at official pricing.", + codeTitle: "Coding tasks", + codeDesc: "Highest accuracy of the three at the lowest per-run cost.", + codeFootnote: "40 coding tasks · averaged over 2 runs · cost estimated at official pricing.", + colFramework: "Framework", + colModel: "Model", + colAccuracy: "Accuracy (%)", + colTokens: "Tokens (M)", + colCost: "Cost ($)", + }, + + features: { + eyebrow: "Features", + title: "The full capability set, one desktop-grade UI", + subtitle: "One-to-one with the web interface's menu — installed means ready.", + items: [ + { + title: "Multi-session chat", + desc: "Any number of sessions per Agent — streaming output, tool approvals and image paste out of the box.", + }, + { + title: "Agent hub", + desc: "Create and manage Agents in one click; names, descriptions and prompts stay editable.", + }, + { + title: "Skill library", + desc: "Browse, install and quick-invoke Skills — Agents can write and optimize their own.", + }, + { + title: "Scheduled tasks", + desc: "Cron-style schedules run Agents on time, fully traced, unattended.", + }, + { + title: "Subagents", + desc: "Delegate work to parallel Subagents — independent and isolated from each other.", + }, + { + title: "Cost center", + desc: "Daily trends for Tokens, requests and cost, with per-model success rates and anomalies.", + }, + { + title: "Trace view", + desc: "Replay every request and tool call round by round, with Token breakdown and timing.", + }, + { + title: "Agent evaluation", + desc: "Built-in Benchmark suites and scoreboards — scores keep climbing as Agents evolve.", + }, + { + title: "Multi-user management", + desc: "Admins provision users; each gets an independent Project with isolated data.", + }, + ], + }, + + security: { + eyebrow: "Security", + title: "Evolution within bounds, data within walls", + subtitle: "A runtime boundary designed for enterprise data security.", + items: [ + { + title: "Open source, local deployment", + desc: "A fully auditable open-source kernel; data lives in local directories and never passes through third-party services.", + }, + { + title: "Bounded evolution", + desc: "Self-improvement is strictly confined to Workspace and Skills — the harness core security boundary is never modified.", + }, + { + title: "Approvals & audit", + desc: "Tool calls require user approval first, and every decision is written to the Trace as an audit event.", + }, + { + title: "Credential isolation", + desc: "Credentials land as hidden 0600 files, are barred from the system prompt, and stay masked throughout the UI.", + }, + ], + }, + + cta: { + title: "Complex AI development, made ever simpler", + subtitle: + "Through continuous evolution, PenguinHarness gives you a more efficient, more reliable, lower-hallucination and lower-cost Agent productivity engine.", + install: "Install now", + docs: "Read the docs", + }, + + footer: { + tagline: "Efficient Self-Improving Harness for Everyone.", + product: "Product", + resources: "Resources", + quickstart: "Quick start", + features: "Features", + benchmark: "Benchmark", + blog: "Blog", + repo: "GitHub repository", + docs: "Documentation", + releases: "Releases", + license: "Apache-2.0 License", + copyright: "© 2026 Prism Shadow · Open source under Apache-2.0", + }, + + blog: { + title: "Blog", + subtitle: "Product news and release notes", + all: "All", + news: "Product news", + changelog: "Release notes", + back: "Back to blog", + empty: "No posts in this category yet", + notFound: "Post not found", + backHome: "Back to home", + toc: "On this page", + }, +}; diff --git a/packages/landing/src/lib/strings.ts b/packages/landing/src/lib/strings.ts new file mode 100644 index 0000000..590c887 --- /dev/null +++ b/packages/landing/src/lib/strings.ts @@ -0,0 +1,361 @@ +/** + * Landing copy (bilingual): this file holds the Chinese dictionary `zh` and the runtime + * active dictionary `S`; the English dictionary lives in strings-en.ts (constrained to + * the same shape by the `Strings` type). Locale switching is handled by state/locale.tsx, + * which calls `setActiveStrings` and remounts the tree keyed by locale — keep `S.x` + * reads inside components. Keep domain terms in standard English casing — Agent, + * Workspace, Token, Task, Skill, Trace, etc. + */ +export const zh = { + siteName: "PenguinHarness", + + nav: { + highlights: "特色", + quickstart: "快速开始", + benchmark: "评测", + contract: "CONTRACT.md", + features: "功能", + blog: "博客", + docs: "文档", + github: "GitHub", + openMenu: "打开菜单", + closeMenu: "关闭菜单", + }, + + theme: { + label: "主题", + light: "浅色", + dark: "深色", + system: "跟随系统", + }, + + lang: { + label: "语言", + zh: "中文", + en: "English", + system: "跟随系统", + }, + + hero: { + badge: "让 Agent 为你构建 Agent", + /** + * Rotating headline: {titlePrefix}{word}{titleSuffix}{titleSuffixNoWrap} — the + * word cycles through titleWords with a gaussian-blur crossfade; titleSuffixNoWrap + * renders as an unbreakable span so CJK line-breaking lands on the phrase boundary. + */ + titlePrefix: "专为", + titleWords: ["开发者", "企业"], + titleSuffix: "设计的", + titleSuffixNoWrap: "高效自进化 Harness", + keywords: ["轻量", "高效", "开源"], + ctaPrimary: "快速开始", + ctaGithub: "GitHub", + installHint: "一行命令安装(Linux / macOS,x64 / arm64,内嵌 Node 运行时,解压即用)", + stats: [ + { value: "1000+", label: "支持模型数量" }, + { value: "1×CPU", label: "最低运行配置" }, + { value: "100%", label: "开源,可本地部署" }, + { value: "首个原生", label: "递归自我进化 Harness" }, + ], + }, + + copy: { + copy: "复制", + copied: "已复制", + }, + + pillars: { + eyebrow: "三大特色", + title: "为构建与进化 Agent 而生", + subtitle: "PenguinHarness 率先把「Agent 构建 Agent」与「递归自我进化」带入开源 Harness。", + root: "PenguinHarness", + concepts: ["Penguin Message", "Penguin SDK", "Penguin Skills"], + diagramLabel: + "PenguinHarness 辐射出 Penguin Message、Penguin SDK 与 Penguin Skills,分别延展出三大特色", + items: [ + { + title: "Simplest Is the Best", + tag: "", + desc: "坚持最小化工具集与简洁的底层接口,以更少的工具调用与 Token 消耗,高效完成复杂任务。", + }, + { + title: "Harness for Building Agents", + tag: "", + desc: "通过 PenguinHarness SDK,让 Agent 从零自主完成 Agent 应用的构建。", + }, + { + title: "Harness for Recursive Self-Improvement", + tag: "", + desc: "通过 PenguinHarness Skills,Agent 以自我评估与自我优化实现递归式自我提升。", + }, + ], + }, + + selfImprove: { + eyebrow: "自我提升循环", + title: "多 Agent 协作,进化自动发生", + subtitle: + "Optimizer 组织多个 Evaluator 为 Target Agent 并行打分,依据分数与运行轨迹定位失分原因,把 Agent 从版本 N 优化到版本 N+1——每一轮都有快照,随时可回退。", + nodeOptimizer: "Optimizer", + nodeEvaluator: "Evaluator × N", + nodeTarget: "Target Agent", + badgeOld: "vN", + badgeNew: "vN+1", + edgeSpawn: "启动并行评测", + edgeBench: "运行 Benchmark", + edgeFeedback: "分数与轨迹", + edgeImprove: "更新提示词与 Skill", + trends: [ + { label: "分数", hint: "不断上升" }, + { label: "成本", hint: "不断降低" }, + { label: "耗时", hint: "不断减少" }, + ], + diagramLabel: + "自我提升循环示意:Optimizer 组织多个 Evaluator 打分,依据分数与轨迹把 Target Agent 从 vN 优化到 vN+1", + }, + + quickstart: { + eyebrow: "快速开始", + title: "三步跑通第一个任务", + subtitle: + "一行命令安装,打开桌面级界面即可让 Agent 开始工作;数据全部保存在本地 ~/.penguin/data 目录。", + step1: "安装", + step1Desc: + "Linux / macOS(x64 / arm64),产物内嵌 Node 运行时,解压即用;升级与重装不触碰数据。", + tabWeb: "Web 界面", + tabCli: "命令行", + webStep2: "启动 Web 界面", + webStep2Desc: + "penguin web 启动本地服务并打开浏览器,用内置管理员 admin / admin123 登录(登录后请尽快修改密码)。", + webCmd: "penguin web # 打开 http://127.0.0.1:7364", + webStep3: "在界面里配置模型,开始对话", + webStep3Desc: + "进入「模型仓库」页,在 DeepSeek 或 OpenRouter 分组里粘贴 API key 并设为默认;回到对话页把第一个任务交给 Agent,例如「分析 data.csv,输出各季度销售额汇总」。", + getKeyPrefix: "获取 API key:", + getDeepseekKey: "DeepSeek 控制台", + getOpenrouterKey: "OpenRouter 控制台", + cliStep2: "配置模型", + cliStep2Desc: "以 DeepSeek 官方 API 或 OpenRouter 网关为例,一条命令完成配置并设为默认。", + tabDeepseek: "DeepSeek", + tabOpenrouter: "OpenRouter", + deepseekCmd: `penguin config model add \\ + --model-id deepseek-v4-pro \\ + --api-key sk-your-deepseek-key \\ + --set-default`, + deepseekNote: "provider 自动推断为 deepseek;省略 --api-key 时回退环境变量 DEEPSEEK_API_KEY。", + openrouterCmd: `penguin config model add \\ + --provider openrouter \\ + --model-id deepseek/deepseek-v4-pro \\ + --api-key sk-or-your-key \\ + --set-default`, + openrouterNote: "网关分组自动预填 OpenAI 兼容协议与 base URL,一个 key 即可访问上千种模型。", + cliStep3: "运行", + cliStep3Desc: "penguin run 直接执行单个任务;penguin chat 进入交互式 REPL。", + runCmd: `penguin run --approve allow-all \\ + --message "分析 data.csv,输出各季度销售额汇总"`, + }, + + showcase: { + eyebrow: "使用场景", + title: "日常任务 + 零代码 AI 研发", + subtitle: + "把重复琐事交给 Agent 全天候自主执行;不写一行代码,从一句需求到可运行的 Agent 应用。", + tagChat: "零代码 AI 研发", + captionChat: "对话即开发:Agent 从零构建一个 Agent 应用,工具真实执行", + tagTraces: "日常任务", + captionTraces: "自主执行全程可回放:每次请求与工具调用、用量与耗时全量可查", + tagBenchmark: "持续进化", + captionBenchmark: "内建 Benchmark 记分板,分数随进化不断上升", + }, + + contract: { + eyebrow: "稳定进化的契约", + title: "CONTRACT.md", + subtitle: "PenguinHarness 以这份契约作为进化的边界和基石:能力可以生长,边界永不漂移。", + intro: "进化需要边界。契约是 Harness 与 Agent 之间的约定:能力生长于内,边界固守于外。", + items: [ + { + term: "工作边界", + text: "所有 Agent 都运行在同一个 Harness 之上:Agent 之下创建 Session,Session 之中执行 Task;自我进化只发生在 Workspace 与 Skill 之内,Harness 内核与安全机制始终不变。", + }, + { + term: "可编辑文件", + text: "Agent 的提示词、技能与配置都是磁盘上可编辑的文件,而不是写死在代码里的常量。你能看到的,Agent 才能改进;你能修改的,Agent 也能学会。", + }, + { + term: "全量追踪", + text: "每一次模型请求、每一次工具调用都完整写入 Trace:花了多少 Token、用了多长时间、为什么失败,事后都能逐条回放。", + }, + { + term: "权限审批", + text: "每个工具调用都先经过审批再执行,每次审批决定都留有审计记录,Agent 做过什么一目了然。", + }, + { + term: "版本控制", + text: "每次优化之前,先保存 Agent State 的版本快照。进化失败或效果回退,一步恢复到任何历史版本。", + }, + { + term: "按需加载", + text: "面向模型的内容先给索引、再按需读取正文,不把整库资料一次性塞进上下文——上下文越干净,行为越稳定。", + }, + { + term: "错误处理", + text: "错误分为可重试与不可重试:可重试的自动重试,不可重试的收敛为消息回给模型,任务不因一次失败而终止。", + }, + { + term: "密钥隔离", + text: "API key 等 credential 存放在隐藏文件里、只经系统接口读写,永远不进入模型上下文,也不以明文出现在界面上。", + }, + { + term: "模型解耦", + text: "模型与 Agent 互不绑定:随时换用更强或更便宜的模型,不需要改动 Agent 本身。", + }, + { + term: "执行轨迹可恢复", + text: "任何 Session 都能从 Trace 完整恢复:进程重启、机器迁移,上下文都不会丢。", + }, + ], + outro: "契约不约束能力的上限,只约束进化的方式。", + }, + + benchmark: { + eyebrow: "Benchmark", + title: "同一模型,效果同级或更好,消耗更低", + subtitle: + "全部使用同一 DeepSeek V4 Pro 模型,与 Claude Code、OpenAI Codex 在两套题库上正面对比。", + higherBetter: "越高越好", + lowerBetter: "越低越好", + dimScore: "准确率", + dimTokens: "Token 用量", + dimCost: "成本", + dataTitle: "复杂数据分析", + dataDesc: "与 Claude Code 准确率持平、显著超过 OpenAI Codex,同时 Token 与成本更低。", + dataFootnote: "15 道复杂数据分析任务 · 1 次运行取均值 · 成本按官方计价估算。", + codeTitle: "代码任务", + codeDesc: "三者中准确率最高、单次成本最低。", + codeFootnote: "40 道代码任务 · 2 次运行取均值 · 成本按官方计价估算。", + colFramework: "实验框架", + colModel: "模型名称", + colAccuracy: "准确率(%)", + colTokens: "Token 用量(M)", + colCost: "成本($)", + }, + + features: { + eyebrow: "主要功能", + title: "桌面级界面里的完整能力", + subtitle: "与 Web 界面的菜单一一对应,装好即用。", + items: [ + { + title: "多 Session 会话", + desc: "每个 Agent 可开任意多个会话,流式输出、工具审批与图片粘贴开箱即用。", + }, + { + title: "智能体仓库", + desc: "一键创建与管理多个 Agent,名称、描述与提示词随时可改。", + }, + { + title: "技能库", + desc: "浏览、安装、快捷调用 Skill,Agent 也能编写并优化自己的技能。", + }, + { + title: "定时任务", + desc: "计划调度让 Agent 到点自动执行,全程留痕,无人值守。", + }, + { + title: "子 Agent", + desc: "任务可委派给 Subagent 并行协作,各自独立、互不干扰。", + }, + { + title: "成本中心", + desc: "Token、请求与成本逐日趋势,各模型成功率与异常明细一目了然。", + }, + { + title: "轨迹观测", + desc: "逐轮回放每次请求与工具调用,Token 细分与耗时全量可查。", + }, + { + title: "Agent 评估", + desc: "内建 Benchmark 题库与记分板,分数随进化持续上升。", + }, + { + title: "多用户管理", + desc: "管理员创建用户,各自拥有独立 Project,数据相互隔离。", + }, + ], + }, + + security: { + eyebrow: "安全", + title: "进化不越界,数据不出域", + subtitle: "为企业级数据安全而设计的运行边界。", + items: [ + { + title: "开源,本地部署", + desc: "内核完全开源可审计,数据保存在本地目录,不经过任何第三方服务。", + }, + { + title: "进化范围受限", + desc: "自我进化严格限制在 Workspace 与 Skill 内,不修改 Harness 核心安全边界。", + }, + { + title: "权限审批与审计", + desc: "工具调用先经用户批准,审批结果全部写入 Trace 审计事件。", + }, + { + title: "密钥隔离", + desc: "credential 以 0600 隐藏文件落盘,系统 Prompt 禁读,界面全程掩码。", + }, + ], + }, + + cta: { + title: "让复杂的 AI 开发越来越简单", + subtitle: + "通过不断进化,PenguinHarness 为你提供更高效、更可靠、更低幻觉、更低成本的 Agent 生产力引擎。", + install: "立即安装", + docs: "阅读文档", + }, + + footer: { + tagline: "Efficient Self-Improving Harness for Everyone.", + product: "产品", + resources: "资源", + quickstart: "快速开始", + features: "功能", + benchmark: "评测", + blog: "博客", + repo: "GitHub 仓库", + docs: "文档", + releases: "Releases", + license: "Apache-2.0 License", + copyright: "© 2026 Prism Shadow · 基于 Apache-2.0 协议开源", + }, + + blog: { + title: "博客", + subtitle: "产品动态与更新日志", + all: "全部", + news: "产品动态", + changelog: "更新日志", + back: "返回博客", + empty: "该分类下暂无文章", + notFound: "文章不存在", + backHome: "返回首页", + toc: "目录", + }, +}; + +/** Dictionary shape (constrains the English dictionary so keys line up). */ +export type Strings = typeof zh; + +/** + * Runtime active dictionary (live binding): the locale Provider calls setActiveStrings + * to switch before render, and remounts the whole tree keyed by locale so every `S.x` + * read reflects the current language. + */ +export let S: Strings = zh; + +export function setActiveStrings(next: Strings): void { + S = next; +} diff --git a/packages/landing/src/lib/toc.ts b/packages/landing/src/lib/toc.ts new file mode 100644 index 0000000..97a6510 --- /dev/null +++ b/packages/landing/src/lib/toc.ts @@ -0,0 +1,41 @@ +/** + * Blog table-of-contents helpers: extract ##/### headings from a Markdown body + * (skipping fenced code blocks) and slugify them the same way the rendered + * headings do, so TOC anchors and heading ids always match. Pure and unit-testable. + */ + +export interface TocEntry { + id: string; + text: string; + depth: 2 | 3; +} + +/** Heading text -> anchor id (keeps CJK, lowercases latin, hyphenates spaces). */ +export function slugifyHeading(text: string): string { + return text + .trim() + .toLowerCase() + .replace(/[^\p{L}\p{N}\s-]/gu, "") + .replace(/\s+/g, "-"); +} + +export function extractToc(body: string): TocEntry[] { + const entries: TocEntry[] = []; + let inFence = false; + for (const line of body.split("\n")) { + if (/^\s*(```|~~~)/.test(line)) { + inFence = !inFence; + continue; + } + if (inFence) continue; + const match = /^(#{2,3})\s+(.+?)\s*$/.exec(line); + if (!match) continue; + const text = match[2]!; + entries.push({ + id: slugifyHeading(text), + text, + depth: match[1]!.length === 2 ? 2 : 3, + }); + } + return entries; +} diff --git a/packages/landing/src/main.tsx b/packages/landing/src/main.tsx new file mode 100644 index 0000000..bbfc1dd --- /dev/null +++ b/packages/landing/src/main.tsx @@ -0,0 +1,14 @@ +/** Landing entry point: mounts the React root component. */ +import { StrictMode } from "react"; +import { createRoot } from "react-dom/client"; +import { App } from "./app"; +import "./styles.css"; + +const container = document.getElementById("root"); +if (!container) throw new Error("#root mount point not found"); + +createRoot(container).render( + + + , +); diff --git a/packages/landing/src/pages/blog-list.tsx b/packages/landing/src/pages/blog-list.tsx new file mode 100644 index 0000000..29d7f4b --- /dev/null +++ b/packages/landing/src/pages/blog-list.tsx @@ -0,0 +1,81 @@ +/** Blog list: category chips (all / product news / release notes) + post cards. */ +import { useState } from "react"; +import { Link } from "react-router"; +import { S } from "../lib/strings"; +import { useLocale } from "../state/locale"; +import { postsFor } from "../lib/blog"; +import type { BlogCategory } from "../lib/blog"; +import { CategoryBadge } from "../components/category-badge"; + +type Filter = "all" | BlogCategory; + +export function BlogListPage() { + const { locale } = useLocale(); + const [filter, setFilter] = useState("all"); + const posts = postsFor(locale, filter === "all" ? undefined : filter); + + const chip = (active: boolean) => + `rounded-full border px-3.5 py-1.5 text-sm transition-colors ${ + active + ? "border-gray-900 bg-gray-900 text-white dark:border-gray-100 dark:bg-gray-100 dark:text-gray-900" + : "border-gray-200 text-gray-600 hover:bg-gray-50 dark:border-gray-800 dark:text-gray-400 dark:hover:bg-gray-900" + }`; + + return ( +
+
+

{S.blog.title}

+

{S.blog.subtitle}

+
+ + + +
+
+ +
+ {posts.length === 0 && ( +

+ {S.blog.empty} +

+ )} + {posts.map((post) => ( + +
+ + +
+

+ {post.title} +

+ {post.excerpt && ( +

+ {post.excerpt} +

+ )} + + ))} +
+
+ ); +} diff --git a/packages/landing/src/pages/blog-post.tsx b/packages/landing/src/pages/blog-post.tsx new file mode 100644 index 0000000..51b5e90 --- /dev/null +++ b/packages/landing/src/pages/blog-post.tsx @@ -0,0 +1,164 @@ +/** + * Blog post: renders the local Markdown body (react-markdown + GFM) in .md-body + * style, with a sticky table of contents on wide screens. Headings get slug ids + * (same slugifier as the TOC) and the active section is tracked with an + * IntersectionObserver so the TOC highlights while scrolling. + */ +import { useEffect, useMemo, useState } from "react"; +import type { ReactNode } from "react"; +import Markdown from "react-markdown"; +import remarkGfm from "remark-gfm"; +import { Link, useParams } from "react-router"; +import { S } from "../lib/strings"; +import { useLocale } from "../state/locale"; +import { getPost } from "../lib/blog"; +import { extractToc, slugifyHeading } from "../lib/toc"; +import { CategoryBadge } from "../components/category-badge"; +import { ArrowRightIcon } from "../components/icons"; + +/** Flatten react-markdown heading children to plain text for slugging. */ +function nodeText(node: ReactNode): string { + if (node === null || node === undefined || typeof node === "boolean") return ""; + if (typeof node === "string" || typeof node === "number") return String(node); + if (Array.isArray(node)) return node.map(nodeText).join(""); + if (typeof node === "object" && "props" in node) { + return nodeText((node as { props: { children?: ReactNode } }).props.children); + } + return ""; +} + +function Toc({ entries, activeId }: { entries: ReturnType; activeId: string }) { + return ( + + ); +} + +export function BlogPostPage() { + const { slug = "" } = useParams(); + const { locale } = useLocale(); + const post = getPost(slug, locale); + const toc = useMemo(() => (post ? extractToc(post.body) : []), [post]); + const [activeId, setActiveId] = useState(""); + + // Track the last heading scrolled past the reading line (~100px under the sticky + // header) for TOC highlighting. Scroll-position based rather than an + // IntersectionObserver: with an observer nothing intersects the narrow top band + // between headings, so the highlight would stall or skip while scrolling. + useEffect(() => { + if (toc.length < 2) return; + let raf = 0; + const update = () => { + raf = 0; + const doc = document.documentElement; + const atBottom = window.innerHeight + window.scrollY >= doc.scrollHeight - 4; + let current = toc[0]!.id; + if (atBottom) { + current = toc[toc.length - 1]!.id; + } else { + for (const entry of toc) { + const el = document.getElementById(entry.id); + if (el && el.getBoundingClientRect().top <= 100) current = entry.id; + } + } + setActiveId(current); + }; + const onScroll = () => { + if (raf === 0) raf = requestAnimationFrame(update); + }; + update(); + window.addEventListener("scroll", onScroll, { passive: true }); + window.addEventListener("resize", onScroll); + return () => { + window.removeEventListener("scroll", onScroll); + window.removeEventListener("resize", onScroll); + if (raf) cancelAnimationFrame(raf); + }; + }, [toc]); + + if (!post) { + return ( +
+

{S.blog.notFound}

+ + {S.blog.back} + +
+ ); + } + + const showToc = toc.length >= 2; + + return ( +
+
+ + + {S.blog.back} + +
+
+ + +
+

{post.title}

+
+
+ ( +

+ {children} +

+ ), + h3: ({ children }) => ( +

+ {children} +

+ ), + }} + > + {post.body} +
+
+
+ {showToc && } +
+ ); +} diff --git a/packages/landing/src/pages/home.tsx b/packages/landing/src/pages/home.tsx new file mode 100644 index 0000000..2ef202c --- /dev/null +++ b/packages/landing/src/pages/home.tsx @@ -0,0 +1,28 @@ +/** Home: section composition, ordered to match the nav anchors. */ +import { Hero } from "../sections/hero"; +import { Pillars } from "../sections/pillars"; +import { SelfImprove } from "../sections/self-improve"; +import { Quickstart } from "../sections/quickstart"; +import { Showcase } from "../sections/showcase"; +import { Benchmark } from "../sections/benchmark"; +import { Contract } from "../sections/contract"; +import { Features } from "../sections/features"; +import { Security } from "../sections/security"; +import { Cta } from "../sections/cta"; + +export function HomePage() { + return ( + <> + + + + + + + + + + + + ); +} diff --git a/packages/landing/src/router.tsx b/packages/landing/src/router.tsx new file mode 100644 index 0000000..fdc49d6 --- /dev/null +++ b/packages/landing/src/router.tsx @@ -0,0 +1,84 @@ +/** + * Router: home + blog list + blog post inside a shared Nav/Footer layout. + * basename comes from Vite's BASE_URL so the site works under the GitHub Pages + * project subpath; scroll restores to top on route change (hash targets excluded). + */ +import { useEffect } from "react"; +import { BrowserRouter, Navigate, Outlet, Route, Routes, useLocation } from "react-router"; +import { DOCS_URL } from "./lib/links"; +import { Nav } from "./components/nav"; +import { Footer } from "./components/footer"; +import { NeonBackground } from "./components/neon-bg"; +import { HomePage } from "./pages/home"; +import { BlogListPage } from "./pages/blog-list"; +import { BlogPostPage } from "./pages/blog-post"; + +/** + * Deep links into the docs SPA that missed a real file (e.g. an unknown slug) land on + * the site-root 404.html, which boots this landing SPA. Hand them to the docs index — + * unless we are already at the docs index (landing dev serves index.html for every + * path, so redirecting to the same URL would reload forever). + */ +function DocsRedirect() { + const target = new URL(DOCS_URL, window.location.origin).pathname; + if (window.location.pathname !== target) { + window.location.replace(target); + return null; + } + return ; +} + +/** + * Last history entry whose scroll was already handled. Module-level so it survives + * the locale-keyed remount of the whole tree: switching language re-mounts Layout, + * and without this guard the navigation effect would re-run (jumping to the hash or + * to the top) and defeat LocaleScope's scroll preservation. + */ +let handledLocationKey = ""; + +function Layout() { + const { pathname, hash, key } = useLocation(); + // Genuine route change: jump to top; with a hash (e.g. /#quickstart after leaving + // the blog) scroll to the target once the section is in the DOM. + useEffect(() => { + if (key === handledLocationKey) return; + handledLocationKey = key; + if (hash) { + const el = document.getElementById(hash.slice(1)); + if (el) { + el.scrollIntoView(); + return; + } + } + window.scrollTo(0, 0); + }, [pathname, hash, key]); + + return ( +
+ +
+ ); +} + +export function AppRouter() { + const basename = import.meta.env.BASE_URL.replace(/\/+$/, "") || "/"; + return ( + + + }> + } /> + } /> + } /> + } /> + + {/* Outside Layout: a pure redirect, no nav/footer flash. */} + } /> + + + ); +} diff --git a/packages/landing/src/sections/benchmark.tsx b/packages/landing/src/sections/benchmark.tsx new file mode 100644 index 0000000..223bee9 --- /dev/null +++ b/packages/landing/src/sections/benchmark.tsx @@ -0,0 +1,268 @@ +/** + * Benchmark section: two suites (complex data analysis + coding tasks), same model + * everywhere, rendered in one identical format — three small multiples (accuracy / + * Tokens / cost, one measure per axis) plus a five-column table of per-run means + * (framework / model / accuracy % / Tokens M / cost $). Suite specifics (case count, + * runs, thinking level, timeout, pricing) sit in a small footnote under each table. + * Emphasis form: PenguinHarness wears the brand hue, competitors the de-emphasis + * gray; per-bar identity comes from logo+name labels, every cap is value-labeled, + * and the exact table relieves the sub-3:1 gray fills. + */ +import { S } from "../lib/strings"; +import { Section } from "../components/section"; +import { HarnessLogo } from "../components/harness-logo"; +import { + CODE_BENCH, + DATA_BENCH, + formatAccuracy, + formatPct, + formatTokensM, + formatUsd, +} from "../lib/benchmark-data"; +import type { BenchResult } from "../lib/benchmark-data"; + +/** Bar path: 4px rounded data-end, square at the baseline. */ +function barPath(x: number, w: number, yTop: number, yBase: number): string { + const r = Math.min(4, Math.max(0, yBase - yTop)); + return [ + `M${x},${yBase}`, + `V${yTop + r}`, + `Q${x},${yTop} ${x + r},${yTop}`, + `H${x + w - r}`, + `Q${x + w},${yTop} ${x + w},${yTop + r}`, + `V${yBase}`, + "Z", + ].join(""); +} + +const VIEW_W = 320; +const VIEW_H = 210; +const PLOT_TOP = 28; +const BASELINE = 168; +const BAR_W = 24; + +function BarPanel({ + title, + hint, + rows, + values, + format, +}: { + title: string; + hint: string; + rows: BenchResult[]; + values: number[]; + format: (v: number) => string; +}) { + const slot = VIEW_W / rows.length; + // Zoomed domain: the baseline is NOT forced to zero — it sits below the smallest + // value by ~60% of the data range, so differences between bars stay visible. The + // truncation is disclosed by the baseline value label at the axis start. + const dataMax = Math.max(...values); + const dataMin = Math.min(...values); + const range = dataMax - dataMin || dataMax || 1; + const lo = Math.max(0, dataMin - range * 0.6); + const hi = dataMax + range * 0.25; + return ( +
+
+ {title} + {hint} +
+ `${r.framework} ${format(values[i] ?? 0)}`).join(", ")}`} + > + {rows.map((row, i) => { + const v = values[i] ?? 0; + const h = ((v - lo) / (hi - lo)) * (BASELINE - PLOT_TOP); + const yTop = BASELINE - h; + const cx = slot * i + slot / 2; + return ( + + {`${row.framework} · ${format(v)}`} + + {/* Value on the cap — text tokens, never the series color. */} + + {format(v)} + + {/* Identity label: brand mark + name, centered under the bar. */} + +
+ + {row.framework} +
+
+
+ ); + })} + + {/* Baseline value: discloses that the axis starts above zero. */} + + {format(lo)} + +
+
+ ); +} + +const TH = "px-3 py-2.5 font-medium whitespace-nowrap"; + +/** + * One suite, one format: header row + three panels + the unified five-column table + * + a small footnote carrying the run settings. Only numbers and names differ + * between suites (and the decimal precision they were published at). + */ +function SuiteBlock({ + title, + desc, + rows, + accDp, + tokenDp, + costDp, + footnote, +}: { + title: string; + desc: string; + rows: BenchResult[]; + accDp: number; + tokenDp: number; + costDp: number; + footnote: string; +}) { + const rowCls = (emphasized?: boolean) => + `border-b border-gray-100 last:border-0 dark:border-gray-800/60 ${ + emphasized ? "bg-brand-25 dark:bg-brand-950/40" : "" + }`; + return ( +
+
+

{title}

+

{desc}

+
+
+ r.accuracyPct)} + format={formatPct} + /> + r.tokensM)} + format={(v) => formatTokensM(v, tokenDp)} + /> + r.costUsd)} + format={(v) => formatUsd(v, costDp)} + /> +
+
+ + + + + + + + + + + + {rows.map((r) => ( + + + + + + + + ))} + +
{S.benchmark.colFramework}{S.benchmark.colModel}{S.benchmark.colAccuracy}{S.benchmark.colTokens}{S.benchmark.colCost}
+ + + {r.framework} + + + {r.model} + + {formatAccuracy(r.accuracyPct, accDp)} + + {r.tokensM.toFixed(tokenDp)} + {r.costUsd.toFixed(costDp)}
+
+

{footnote}

+
+ ); +} + +export function Benchmark() { + return ( +
+
+ + +
+
+ ); +} diff --git a/packages/landing/src/sections/contract.tsx b/packages/landing/src/sections/contract.tsx new file mode 100644 index 0000000..fb77251 --- /dev/null +++ b/packages/landing/src/sections/contract.tsx @@ -0,0 +1,81 @@ +/** + * CONTRACT.md: the signature section — the "X first" principles that keep + * self-evolution stable, rendered as a Markdown file inside a file-window card + * with a copy button that yields the underlying Markdown verbatim. + * High-level covenant only; implementation detail stays out of the landing page. + */ +import { S } from "../lib/strings"; +import { Section } from "../components/section"; +import { CopyButton } from "../components/copy-button"; + +/** The card rendered as real Markdown (what the copy button places on the clipboard). */ +function contractMarkdown(): string { + return [ + "# CONTRACT.md", + "", + `> ${S.contract.intro}`, + "", + ...S.contract.items.map((item) => `- **${item.term}** — ${item.text}`), + "", + `> ${S.contract.outro}`, + "", + ].join("\n"); +} + +export function Contract() { + return ( +
+
+ {/* File chrome */} +
+
+ + {/* Markdown-flavored body */} +
+

+ # + CONTRACT.md +

+

+ > + {S.contract.intro} +

+
    + {S.contract.items.map((item) => ( +
  • + +

    + + **{item.term}** + + — + {item.text} +

    +
  • + ))} +
+

+ > + {S.contract.outro} +

+
+
+
+ ); +} diff --git a/packages/landing/src/sections/cta.tsx b/packages/landing/src/sections/cta.tsx new file mode 100644 index 0000000..1096826 --- /dev/null +++ b/packages/landing/src/sections/cta.tsx @@ -0,0 +1,35 @@ +/** Closing call-to-action: the productivity-engine pitch + install / docs buttons. */ +import { S } from "../lib/strings"; +import { DOCS_URL } from "../lib/links"; +import { Section } from "../components/section"; +import { ArrowRightIcon } from "../components/icons"; + +export function Cta() { + return ( +
+
+

+ {S.cta.title} +

+

+ {S.cta.subtitle} +

+ +
+
+ ); +} diff --git a/packages/landing/src/sections/features.tsx b/packages/landing/src/sections/features.tsx new file mode 100644 index 0000000..54ea562 --- /dev/null +++ b/packages/landing/src/sections/features.tsx @@ -0,0 +1,64 @@ +/** + * Feature grid mirroring the Web App's menu: multi-session chat, agent hub, skill + * library, scheduled tasks, subagents, cost center, trace view, agent evaluation, + * multi-user management — icon + title + one-line description. + */ +import { S } from "../lib/strings"; +import { Section } from "../components/section"; +import { + ActivityIcon, + BarChartIcon, + BotIcon, + ClockIcon, + MessageSquareIcon, + PieChartIcon, + ShareIcon, + SparklesIcon, + UsersIcon, +} from "../components/icons"; + +/** Icon order matches S.features.items. */ +const ICONS = [ + MessageSquareIcon, + BotIcon, + SparklesIcon, + ClockIcon, + ShareIcon, + PieChartIcon, + ActivityIcon, + BarChartIcon, + UsersIcon, +]; + +export function Features() { + return ( +
+
+ {S.features.items.map((item, i) => { + const IconCmp = ICONS[i] ?? SparklesIcon; + return ( +
+ {/* Oversized faint icon as the card backdrop (decorative). */} + +

{item.title}

+

+ {item.desc} +

+
+ ); + })} +
+
+ ); +} diff --git a/packages/landing/src/sections/hero.tsx b/packages/landing/src/sections/hero.tsx new file mode 100644 index 0000000..2565b42 --- /dev/null +++ b/packages/landing/src/sections/hero.tsx @@ -0,0 +1,142 @@ +/** + * Hero: enlarged logo + product name, the slogan inside the pill badge, then a + * bilingual headline whose rotating word crossfades through a gaussian blur + * (Developers <-> Enterprises), three keywords, the install one-liner and stats. + * The rotating word is a stacked inline-grid so line width never jumps. + */ +import { Fragment, useEffect, useState } from "react"; +import { S } from "../lib/strings"; +import { INSTALL_CMD, REPO_URL } from "../lib/links"; +import { CopyButton } from "../components/copy-button"; +import { ArrowRightIcon, GitHubIcon } from "../components/icons"; + +const ROTATE_MS = 2600; + +function RotatingWord({ words }: { words: string[] }) { + const [active, setActive] = useState(0); + useEffect(() => { + if (words.length < 2) return; + const timer = setInterval(() => setActive((i) => (i + 1) % words.length), ROTATE_MS); + return () => clearInterval(timer); + }, [words.length]); + return ( + + {words.map((word, i) => ( + + {word} + + ))} + + ); +} + +export function Hero() { + return ( +
+
+ ); +} diff --git a/packages/landing/src/sections/pillars.tsx b/packages/landing/src/sections/pillars.tsx new file mode 100644 index 0000000..8d204eb --- /dev/null +++ b/packages/landing/src/sections/pillars.tsx @@ -0,0 +1,170 @@ +/** + * Three pillars, introduced by a radiating diagram: PenguinHarness fans out into + * Penguin Message / Penguin SDK / Penguin Skills, and each concept extends into one + * pillar card below. Arrow paths carry slow-moving dots (SMIL animateMotion); the + * diagram is hidden on small screens where the three columns stack. + */ +import { S } from "../lib/strings"; +import { Section } from "../components/section"; +import { BotIcon, FeatherIcon, RefreshIcon } from "../components/icons"; + +const ICONS = [FeatherIcon, BotIcon, RefreshIcon]; + +/** Fan paths from the root's bottom edge to each concept chip (also the dot tracks). */ +const FAN_PATHS = [ + "M322,58 C260,92 190,102 132,120", + "M360,58 L360,120", + "M398,58 C460,92 530,102 588,120", +]; +/** Short connectors from each concept chip into its pillar card below. */ +const DROP_PATHS = ["M120,164 L120,198", "M360,164 L360,198", "M600,164 L600,198"]; + +function FlowDot({ path, dur, begin }: { path: string; dur: string; begin: string }) { + // Same color as the arrow strokes so the dot reads as flow on the line, not an accent. + return ( + + {/* Stop at 90% of the path so the dot never rides over the arrowhead. */} + + + ); +} + +function RadialDiagram() { + return ( + + + + + + + + + {FAN_PATHS.map((d) => ( + + ))} + {DROP_PATHS.map((d) => ( + + ))} + + {FAN_PATHS.map((d, i) => ( + + ))} + {DROP_PATHS.map((d, i) => ( + + ))} + + {/* Root node */} + + + + {S.pillars.root} + + + {/* Concept chips */} + {S.pillars.concepts.map((concept, i) => { + const cx = [120, 360, 600][i] ?? 360; + return ( + + + + {concept} + + + ); + })} + + ); +} + +export function Pillars() { + return ( +
+ +
+ {S.pillars.items.map((item, i) => { + const IconCmp = ICONS[i] ?? FeatherIcon; + return ( +
+ {/* Oversized faint icon as the card backdrop (brand-tinted, decorative). */} + +

{item.title}

+

+ {item.desc} +

+
+ ); + })} +
+
+ ); +} diff --git a/packages/landing/src/sections/quickstart.tsx b/packages/landing/src/sections/quickstart.tsx new file mode 100644 index 0000000..5bc257c --- /dev/null +++ b/packages/landing/src/sections/quickstart.tsx @@ -0,0 +1,212 @@ +/** + * Quick start: install, then a Web UI / CLI tab pair (Web is the default and never + * touches the command line beyond `penguin web` — models are configured inside the + * interface). API-key console links open in a new tab. Localized commands live in + * the string dictionaries. + */ +import { useState } from "react"; +import type { ReactNode } from "react"; +import { S } from "../lib/strings"; +import { DEEPSEEK_KEYS_URL, INSTALL_CMD, OPENROUTER_KEYS_URL } from "../lib/links"; +import { Section } from "../components/section"; +import { CodeCard } from "../components/code-card"; +import { + DownloadIcon, + ExternalLinkIcon, + MonitorIcon, + PlayIcon, + SlidersIcon, + TerminalIcon, +} from "../components/icons"; + +function Step({ + index, + icon, + title, + desc, + children, +}: { + index: number; + icon: ReactNode; + title: string; + desc: string; + children?: ReactNode; +}) { + return ( +
  • + + {icon} + +

    + {index}. + {title} +

    +

    {desc}

    + {children &&
    {children}
    } +
  • + ); +} + +/** API-key console links (open in a new tab). */ +function KeyLinks() { + const link = + "inline-flex items-center gap-1 text-brand-700 underline decoration-brand-300 underline-offset-2 transition-colors hover:text-brand-600 dark:text-brand-300 dark:decoration-brand-700"; + return ( +

    + {S.quickstart.getKeyPrefix} + + {S.quickstart.getDeepseekKey} + + + · + + {S.quickstart.getOpenrouterKey} + + +

    + ); +} + +export function Quickstart() { + const [mode, setMode] = useState<"web" | "cli">("web"); + const [provider, setProvider] = useState<"deepseek" | "openrouter">("deepseek"); + + const modeBtn = (active: boolean) => + `inline-flex items-center gap-1.5 rounded-md px-3.5 py-1.5 text-sm font-medium transition-colors ${ + active + ? "bg-gray-900 text-white dark:bg-gray-100 dark:text-gray-900" + : "text-gray-600 hover:bg-gray-100 dark:text-gray-400 dark:hover:bg-gray-800" + }`; + const providerBtn = (active: boolean) => + `rounded-md px-3 py-1.5 text-sm font-medium transition-colors ${ + active + ? "bg-gray-900 text-white dark:bg-gray-100 dark:text-gray-900" + : "text-gray-600 hover:bg-gray-100 dark:text-gray-400 dark:hover:bg-gray-800" + }`; + + return ( +
    +
    +
    + + +
    + +
      + } + title={S.quickstart.step1} + desc={S.quickstart.step1Desc} + > + + + + {mode === "web" ? ( + <> + } + title={S.quickstart.webStep2} + desc={S.quickstart.webStep2Desc} + > + + + } + title={S.quickstart.webStep3} + desc={S.quickstart.webStep3Desc} + > + + + + ) : ( + <> + } + title={S.quickstart.cliStep2} + desc={S.quickstart.cliStep2Desc} + > +
      + + +
      + {provider === "deepseek" ? ( + <> + +

      + {S.quickstart.deepseekNote} +

      + + ) : ( + <> + +

      + {S.quickstart.openrouterNote} +

      + + )} + +
      + } + title={S.quickstart.cliStep3} + desc={S.quickstart.cliStep3Desc} + > + + + + )} +
    +
    +
    + ); +} diff --git a/packages/landing/src/sections/security.tsx b/packages/landing/src/sections/security.tsx new file mode 100644 index 0000000..84a0a75 --- /dev/null +++ b/packages/landing/src/sections/security.tsx @@ -0,0 +1,39 @@ +/** Security strip: the enterprise-grade runtime boundary in four points. */ +import { S } from "../lib/strings"; +import { Section } from "../components/section"; +import { FileCheckIcon, FrameIcon, KeyIcon, ShieldCheckIcon } from "../components/icons"; + +const ICONS = [ShieldCheckIcon, FrameIcon, FileCheckIcon, KeyIcon]; + +export function Security() { + return ( +
    +
    + {S.security.items.map((item, i) => { + const IconCmp = ICONS[i] ?? ShieldCheckIcon; + return ( +
    + {/* Oversized faint icon as the card backdrop (decorative). */} + +

    {item.title}

    +

    + {item.desc} +

    +
    + ); + })} +
    +
    + ); +} diff --git a/packages/landing/src/sections/self-improve.tsx b/packages/landing/src/sections/self-improve.tsx new file mode 100644 index 0000000..ad7272b --- /dev/null +++ b/packages/landing/src/sections/self-improve.tsx @@ -0,0 +1,426 @@ +/** + * Self-improvement loop, centered on the Target Agent's vN -> vN+1 upgrade: the + * Optimizer orchestrates Evaluators to score vN (scores & traces flow back as edge + * labels, not as a node), analyzes them, and ships the upgraded vN+1 — drawn as a + * bold evolve arrow between the two versions. Beside it, three outcome trends with + * real axes and ticks, each line growing from zero rightwards on a slow loop. + * Illustrative; all colors are theme-aware Tailwind fill/stroke classes. + */ +import { S } from "../lib/strings"; +import { Section } from "../components/section"; + +/** Rounded node with a centered label and an optional version pill. */ +function Node({ + x, + y, + w = 150, + h = 48, + label, + kind = "neutral", + badge, + badgeKind = "neutral", +}: { + x: number; + y: number; + w?: number; + h?: number; + label: string; + kind?: "accent" | "target" | "neutral"; + badge?: string; + badgeKind?: "accent" | "neutral"; +}) { + const rect = + kind === "accent" + ? "fill-white stroke-brand-500 dark:fill-gray-900 dark:stroke-brand-400" + : kind === "target" + ? "fill-brand-50 stroke-brand-200 dark:fill-brand-950 dark:stroke-brand-800" + : "fill-gray-50 stroke-gray-300 dark:fill-gray-800 dark:stroke-gray-700"; + const text = + kind === "target" ? "fill-brand-800 dark:fill-brand-200" : "fill-gray-900 dark:fill-gray-100"; + return ( + + + + {label} + + {badge && ( + <> + + + {badge} + + + )} + + ); +} + +function EdgeLabel({ + x, + y, + anchor = "middle", + accent = false, + children, +}: { + x: number; + y: number; + anchor?: "start" | "middle" | "end"; + accent?: boolean; + children: string; +}) { + return ( + + {children} + + ); +} + +function FlowDot({ + path, + dur, + begin, + accent = false, + r = 1.5, +}: { + path: string; + dur: string; + begin: string; + accent?: boolean; + r?: number; +}) { + return ( + + {/* Stop at 90% of the path so the dot never rides over the arrowhead. */} + + + ); +} + +/** Loop edges (top machinery) and the bold evolve chord (bottom). */ +const EDGE_SPAWN = "M330,76 L330,154"; +const EDGE_FEEDBACK = "M390,208 L390,82"; +const EDGE_BENCH_OLD = "M295,208 L163,294"; +const EDGE_BENCH_NEW = "M425,208 L557,294"; +const EDGE_EVOLVE = "M216,324 L504,324"; + +/** + * Outcome trend chart with proper axes and ticks; the line grows from zero + * rightwards (pathLength-normalized draw animation looping in styles.css). + */ +function TrendChart({ + label, + hint, + points, + yTicks, + xTicks, + delay, +}: { + label: string; + hint: string; + points: number[]; + yTicks: number[]; + xTicks: Array<{ at: number; text: string }>; + delay: string; +}) { + const W = 210; + const H = 112; + const M = { left: 34, right: 10, top: 10, bottom: 20 }; + const yMax = Math.max(...yTicks); + const yMin = Math.min(...yTicks); + const px = (i: number) => M.left + (i / (points.length - 1)) * (W - M.left - M.right); + const py = (v: number) => M.top + ((yMax - v) / (yMax - yMin || 1)) * (H - M.top - M.bottom); + const path = points.map((v, i) => `${i === 0 ? "M" : "L"}${px(i)},${py(v)}`).join(" "); + const lastX = px(points.length - 1); + const lastY = py(points[points.length - 1] ?? 0); + const axis = "stroke-gray-200 dark:stroke-gray-800"; + const tickText = "fill-gray-400 text-[8.5px] tabular-nums dark:fill-gray-500"; + + return ( +
    +
    +

    {label}

    +

    {hint}

    +
    + +
    + ); +} + +const TRENDS: Array<{ + points: number[]; + yTicks: number[]; + xTicks: Array<{ at: number; text: string }>; +}> = [ + { + points: [52, 58, 56, 64, 70, 75, 79], + yTicks: [0, 50, 100], + xTicks: [ + { at: 0, text: "v1" }, + { at: 3, text: "v4" }, + { at: 6, text: "v7" }, + ], + }, + { + points: [0.42, 0.38, 0.39, 0.33, 0.3, 0.27, 0.25], + yTicks: [0, 0.25, 0.5], + xTicks: [ + { at: 0, text: "v1" }, + { at: 3, text: "v4" }, + { at: 6, text: "v7" }, + ], + }, + { + points: [115, 107, 109, 98, 92, 87, 83], + yTicks: [0, 60, 120], + xTicks: [ + { at: 0, text: "v1" }, + { at: 3, text: "v4" }, + { at: 6, text: "v7" }, + ], + }, +]; + +export function SelfImprove() { + return ( +
    +
    +
    + + + + + + + + + + + {/* Loop edges: Evaluators benchmark BOTH versions (vN and vN+1). */} + + + + + + + {/* The evolve chord: vN -> vN+1 is the story (the Optimizer's update lands here). */} + + + {/* Moving dots along every arrow. */} + + + + + + + + + {S.selfImprove.edgeSpawn} + + + {S.selfImprove.edgeFeedback} + + {/* Centered between the two benchmark diagonals: it applies to both. */} + + {S.selfImprove.edgeBench} + + + {S.selfImprove.edgeImprove} + + + {/* Evaluator stack (x N) */} + + + + + + + + +
    + +
    + {S.selfImprove.trends.map((t, i) => ( + + ))} +
    +
    +
    + ); +} diff --git a/packages/landing/src/sections/showcase.tsx b/packages/landing/src/sections/showcase.tsx new file mode 100644 index 0000000..fe0332e --- /dev/null +++ b/packages/landing/src/sections/showcase.tsx @@ -0,0 +1,112 @@ +/** + * Use cases: daily tasks + zero-code AI development, illustrated with real + * screenshots captured from the Web App with Playwright. Twelve variants — three + * views (chat building an Agent app / trace view / evaluation center) x two UI + * languages x two themes (WebP) — and the one shown always matches the visitor's + * active locale and theme. Each figure carries a scenario tag. + */ +import { S } from "../lib/strings"; +import { useLocale } from "../state/locale"; +import type { Locale } from "../state/locale"; +import { Section } from "../components/section"; +import { BrowserFrame } from "../components/browser-frame"; +import chatZhLight from "../assets/shots/chat-zh-light.webp"; +import chatZhDark from "../assets/shots/chat-zh-dark.webp"; +import chatEnLight from "../assets/shots/chat-en-light.webp"; +import chatEnDark from "../assets/shots/chat-en-dark.webp"; +import tracesZhLight from "../assets/shots/traces-zh-light.webp"; +import tracesZhDark from "../assets/shots/traces-zh-dark.webp"; +import tracesEnLight from "../assets/shots/traces-en-light.webp"; +import tracesEnDark from "../assets/shots/traces-en-dark.webp"; +import benchmarkZhLight from "../assets/shots/benchmark-zh-light.webp"; +import benchmarkZhDark from "../assets/shots/benchmark-zh-dark.webp"; +import benchmarkEnLight from "../assets/shots/benchmark-en-light.webp"; +import benchmarkEnDark from "../assets/shots/benchmark-en-dark.webp"; + +type ShotSet = Record; + +const SHOTS: Record<"chat" | "traces" | "benchmark", ShotSet> = { + chat: { + zh: { light: chatZhLight, dark: chatZhDark }, + en: { light: chatEnLight, dark: chatEnDark }, + }, + traces: { + zh: { light: tracesZhLight, dark: tracesZhDark }, + en: { light: tracesEnLight, dark: tracesEnDark }, + }, + benchmark: { + zh: { light: benchmarkZhLight, dark: benchmarkZhDark }, + en: { light: benchmarkEnLight, dark: benchmarkEnDark }, + }, +}; + +function ThemedShot({ set, alt }: { set: ShotSet; alt: string }) { + const { locale } = useLocale(); + const pair = set[locale]; + // Explicit intrinsic size (all shots are 1920x1200): reserves layout before the + // image loads, so in-page anchor jumps don't drift when the showcase pops in. + return ( + <> + {alt} + {alt} + + ); +} + +function Caption({ tag, text }: { tag: string; text: string }) { + return ( +
    + + {tag} + + {text} +
    + ); +} + +export function Showcase() { + return ( +
    +
    + + + + +
    + +
    +
    + + + + +
    +
    + + + + +
    +
    +
    + ); +} diff --git a/packages/landing/src/state/locale.tsx b/packages/landing/src/state/locale.tsx new file mode 100644 index 0000000..1c48269 --- /dev/null +++ b/packages/landing/src/state/locale.tsx @@ -0,0 +1,114 @@ +/** + * Language context: zh / en / system (tracks navigator.language). On switch it first + * synchronously calls setActiveStrings, then remounts the tree keyed on locale so every + * `S.x` read reflects the new language; the preference persists to localStorage. + * Same pattern as the Web App. + */ +import { + createContext, + useCallback, + useContext, + useEffect, + useLayoutEffect, + useRef, + useState, +} from "react"; +import type { ReactNode } from "react"; +import { setActiveStrings, zh } from "../lib/strings"; +import { en } from "../lib/strings-en"; + +export type LangPref = "zh" | "en" | "system"; +export type Locale = "zh" | "en"; + +const STORAGE_KEY = "penguin-landing.lang"; + +interface LocaleContextValue { + lang: LangPref; + locale: Locale; + setLang: (lang: LangPref) => void; +} + +const LocaleContext = createContext(null); + +/** Device language -> UI language: zh* -> zh, anything else -> en. */ +export function resolveSystemLocale(language: string | undefined): Locale { + return language?.toLowerCase().startsWith("zh") ? "zh" : "en"; +} + +function systemLocale(): Locale { + return resolveSystemLocale(navigator.language); +} + +function resolve(lang: LangPref): Locale { + return lang === "system" ? systemLocale() : lang; +} + +function initialLang(): LangPref { + const stored = localStorage.getItem(STORAGE_KEY); + if (stored === "zh" || stored === "en" || stored === "system") return stored; + return "system"; +} + +export function LocaleProvider({ children }: { children: ReactNode }) { + const [lang, setLangState] = useState(initialLang); + const [, setSysTick] = useState(0); + + const locale = resolve(lang); + // Switch the active dictionary during render (idempotent): children are keyed on + // locale and render after this component, so they read the post-switch dictionary. + setActiveStrings(locale === "en" ? en : zh); + + // Keep the document language in sync (static index.html ships lang="en"). + useEffect(() => { + document.documentElement.lang = locale === "zh" ? "zh-CN" : "en"; + }, [locale]); + + useEffect(() => { + if (lang !== "system") return; + const onChange = () => setSysTick((t) => t + 1); + window.addEventListener("languagechange", onChange); + return () => window.removeEventListener("languagechange", onChange); + }, [lang]); + + const setLang = useCallback((next: LangPref) => { + localStorage.setItem(STORAGE_KEY, next); + setLangState(next); + }, []); + + return ( + {children} + ); +} + +/** + * Language scope: a remount boundary keyed on locale. The remount briefly empties the + * DOM, which collapses the page height and clamps the scroll position to 0 — so the + * scroll offset is captured during the render that switches locale (old DOM still + * mounted) and restored right after the new tree lays out. + */ +export function LocaleScope({ children }: { children: ReactNode }) { + const { locale } = useLocale(); + const prevLocale = useRef(locale); + const savedScroll = useRef(null); + if (prevLocale.current !== locale) { + prevLocale.current = locale; + savedScroll.current = window.scrollY; + } + useLayoutEffect(() => { + if (savedScroll.current !== null) { + window.scrollTo({ top: savedScroll.current, behavior: "instant" }); + savedScroll.current = null; + } + }, [locale]); + return ( +
    + {children} +
    + ); +} + +export function useLocale(): LocaleContextValue { + const ctx = useContext(LocaleContext); + if (!ctx) throw new Error("useLocale must be used inside LocaleProvider"); + return ctx; +} diff --git a/packages/landing/src/state/theme.tsx b/packages/landing/src/state/theme.tsx new file mode 100644 index 0000000..4f3a8cd --- /dev/null +++ b/packages/landing/src/state/theme.tsx @@ -0,0 +1,63 @@ +/** + * Theme context: light / dark / system (tracks prefers-color-scheme live), toggled + * via the html.dark class + Tailwind dark: variant, persisted to localStorage. + * Same behavior as the Web App, trimmed to what the landing page needs. + */ +import { createContext, useCallback, useContext, useEffect, useState } from "react"; +import type { ReactNode } from "react"; + +export type ThemeMode = "light" | "dark" | "system"; + +const MODE_KEY = "penguin-landing.theme"; + +interface ThemeContextValue { + mode: ThemeMode; + /** Resolved effective theme (system mode resolved against the OS preference). */ + dark: boolean; + setMode: (mode: ThemeMode) => void; +} + +const ThemeContext = createContext(null); + +function initialMode(): ThemeMode { + const stored = localStorage.getItem(MODE_KEY); + if (stored === "light" || stored === "dark" || stored === "system") return stored; + return "system"; +} + +function systemDark(): boolean { + return window.matchMedia("(prefers-color-scheme: dark)").matches; +} + +export function ThemeProvider({ children }: { children: ReactNode }) { + const [mode, setModeState] = useState(initialMode); + const [sysDark, setSysDark] = useState(systemDark); + + const dark = mode === "system" ? sysDark : mode === "dark"; + + useEffect(() => { + document.documentElement.classList.toggle("dark", dark); + }, [dark]); + + useEffect(() => { + if (mode !== "system") return; + const mq = window.matchMedia("(prefers-color-scheme: dark)"); + const onChange = (e: MediaQueryListEvent) => setSysDark(e.matches); + setSysDark(mq.matches); + mq.addEventListener("change", onChange); + return () => mq.removeEventListener("change", onChange); + }, [mode]); + + const setMode = useCallback((next: ThemeMode) => { + localStorage.setItem(MODE_KEY, next); + setModeState(next); + }, []); + + return {children}; +} + +export function useTheme(): ThemeContextValue { + const ctx = useContext(ThemeContext); + if (!ctx) throw new Error("useTheme must be used inside ThemeProvider"); + return ctx; +} diff --git a/packages/landing/src/styles.css b/packages/landing/src/styles.css new file mode 100644 index 0000000..aeafbbb --- /dev/null +++ b/packages/landing/src/styles.css @@ -0,0 +1,405 @@ +/** + * Landing page styles: single-file Tailwind CSS 4 entry point, sharing the Web App's + * visual language — GitHub-style simplicity, solid backgrounds + 1px borders + one + * brand blue accent, no glassmorphism. Dark theme is toggled via the html.dark class. + */ +@import "tailwindcss"; + +/* Dark theme toggled via the html.dark class (Tailwind 4 custom variant). */ +@custom-variant dark (&:where(.dark, .dark *)); + +/* Brand color scale (Google blue family), identical to the Web App. */ +@theme { + --color-brand-25: #f8fbff; + --color-brand-50: #e8f0fe; + --color-brand-100: #d2e3fc; + --color-brand-200: #aecbfa; + --color-brand-300: #8ab4f8; + --color-brand-400: #669df6; + --color-brand-500: #4285f4; + --color-brand-600: #1a73e8; + --color-brand-700: #0b57d0; + --color-brand-800: #0842a0; + --color-brand-900: #062e6f; + --color-brand-950: #041e49; +} + +@layer base { + :root { + color-scheme: light; + } + .dark { + color-scheme: dark; + /* Pure-black dark base, matching the Web App (true-neutral gray overrides). */ + --color-gray-950: #000000; + --color-gray-900: #0d0d0d; + --color-gray-800: #1f1f1f; + --color-gray-700: #303030; + } + html { + scroll-behavior: smooth; + } + body { + @apply bg-white text-gray-900 antialiased dark:bg-gray-950 dark:text-gray-100; + font-family: + ui-sans-serif, + system-ui, + -apple-system, + "Segoe UI", + Roboto, + "PingFang SC", + "Microsoft YaHei", + sans-serif; + } + code, + pre, + kbd { + font-family: ui-monospace, SFMono-Regular, Menlo, Consolas, "Liberation Mono", monospace; + } + button:focus-visible, + a:focus-visible, + summary:focus-visible { + outline: 3px solid rgb(107 114 128 / 0.4); + outline-offset: 2px; + } + button:not(:disabled), + [role="button"]:not(:disabled) { + cursor: pointer; + } + ::selection { + background: rgb(0 0 0 / 0.1); + } + .dark ::selection { + background: rgb(255 255 255 / 0.18); + } + * { + scrollbar-width: thin; + scrollbar-color: rgb(60 64 67 / 0.28) transparent; + } + .dark * { + scrollbar-color: rgb(232 234 237 / 0.2) transparent; + } + /* Anchor sections scroll below the sticky header. */ + section[id] { + scroll-margin-top: 4.5rem; + } +} + +/* ---------- Animations (same tone as the Web App: 160-280ms ease, slight offset) ---------- */ + +@keyframes rise-in { + from { + opacity: 0; + transform: translateY(10px) scale(0.99); + } + to { + opacity: 1; + transform: none; + } +} + +@keyframes fade-in { + from { + opacity: 0; + } + to { + opacity: 1; + } +} + +.anim-rise { + animation: rise-in 280ms cubic-bezier(0.2, 0.7, 0.3, 1) both; +} +.anim-fade { + animation: fade-in 120ms ease-out both; +} + +/* Scroll reveal: sections start slightly offset and settle in when they enter the viewport. */ +.reveal { + opacity: 0; + transform: translateY(14px); + transition: + opacity 500ms cubic-bezier(0.2, 0.7, 0.3, 1), + transform 500ms cubic-bezier(0.2, 0.7, 0.3, 1); +} +.reveal-visible { + opacity: 1; + transform: none; +} + +/* Hero backdrop: a faint dot grid fading out downwards — subtle, no gradients on content. */ +.hero-dots { + background-image: radial-gradient(rgb(26 115 232 / 0.14) 1px, transparent 1px); + background-size: 22px 22px; + mask-image: linear-gradient(to bottom, black 0%, transparent 78%); + -webkit-mask-image: linear-gradient(to bottom, black 0%, transparent 78%); +} +.dark .hero-dots { + background-image: radial-gradient(rgb(138 180 248 / 0.16) 1px, transparent 1px); +} + +/* ---------- Ambient neon backdrop: blurred glows wandering across the screen ---------- + Viewport-FIXED (the page background stays put while content scrolls over it) with a + uniform page background behind every section, so nothing jumps at section boundaries. + Each glow wanders freely across the whole screen on its own slow route (vw/vh + amplitudes), soft and blue-family only (enterprise tone; hue drift stays within blues). Reduced-motion freezes them in place. */ +.neon-bg { + position: fixed; + inset: 0; + z-index: -1; + overflow: hidden; + pointer-events: none; +} +.neon-blob { + position: absolute; + width: 48rem; + height: 48rem; + border-radius: 9999px; + opacity: 0.12; + will-change: transform, filter; + filter: blur(88px); + animation: neon-hue 44s linear infinite alternate; +} +.dark .neon-blob { + opacity: 0.16; +} +.neon-blob-a { + top: -12rem; + left: -10rem; + background: radial-gradient(circle, #4285f4 0%, transparent 66%); + animation: + neon-wander-a 44s ease-in-out infinite, + neon-hue 46s linear infinite alternate; +} +.neon-blob-b { + top: 8%; + right: -12rem; + background: radial-gradient(circle, #0ea5e9 0%, transparent 66%); + animation: + neon-wander-b 56s ease-in-out infinite, + neon-hue 52s linear infinite alternate -16s; +} +.neon-blob-c { + bottom: -14rem; + left: 18%; + background: radial-gradient(circle, #06b6d4 0%, transparent 66%); + animation: + neon-wander-c 50s ease-in-out infinite, + neon-hue 58s linear infinite alternate -30s; +} +.neon-blob-d { + top: 42%; + left: -14rem; + background: radial-gradient(circle, #2563eb 0%, transparent 66%); + animation: + neon-wander-d 62s ease-in-out infinite, + neon-hue 48s linear infinite alternate -22s; +} +@keyframes neon-wander-a { + 0%, + 100% { + transform: translate3d(0, 0, 0) scale(1); + } + 30% { + transform: translate3d(48vw, 22vh, 0) scale(1.12); + } + 60% { + transform: translate3d(16vw, 58vh, 0) scale(0.94); + } + 80% { + transform: translate3d(-6vw, 28vh, 0) scale(1.06); + } +} +@keyframes neon-wander-b { + 0%, + 100% { + transform: translate3d(0, 0, 0) scale(1); + } + 25% { + transform: translate3d(-38vw, 32vh, 0) scale(1.08); + } + 55% { + transform: translate3d(-58vw, 6vh, 0) scale(1.16); + } + 80% { + transform: translate3d(-14vw, -6vh, 0) scale(0.96); + } +} +@keyframes neon-wander-c { + 0%, + 100% { + transform: translate3d(0, 0, 0) scale(1); + } + 30% { + transform: translate3d(30vw, -34vh, 0) scale(1.1); + } + 55% { + transform: translate3d(-12vw, -56vh, 0) scale(0.95); + } + 80% { + transform: translate3d(8vw, -16vh, 0) scale(1.05); + } +} +@keyframes neon-wander-d { + 0%, + 100% { + transform: translate3d(0, 0, 0) scale(1); + } + 25% { + transform: translate3d(34vw, -18vh, 0) scale(1.1); + } + 50% { + transform: translate3d(64vw, 8vh, 0) scale(0.95); + } + 75% { + transform: translate3d(28vw, 24vh, 0) scale(1.08); + } +} +@keyframes neon-hue { + from { + filter: blur(88px) hue-rotate(-8deg); + } + to { + filter: blur(88px) hue-rotate(12deg); + } +} + +/* ---------- Trend-line growth: draw from zero rightwards, hold, fade, loop ---------- + Paths must set pathLength=1; the dot pops in once the line has fully grown. */ +.anim-line-draw { + stroke-dasharray: 1; + stroke-dashoffset: 1; + animation: line-draw var(--draw-dur, 4.6s) cubic-bezier(0.3, 0, 0.35, 1) infinite; +} +.anim-line-dot { + opacity: 0; + animation: line-dot var(--draw-dur, 4.6s) linear infinite; +} +@keyframes line-draw { + 0% { + stroke-dashoffset: 1; + opacity: 1; + } + 55% { + stroke-dashoffset: 0; + opacity: 1; + } + 84% { + stroke-dashoffset: 0; + opacity: 1; + } + 94%, + 100% { + stroke-dashoffset: 0; + opacity: 0; + } +} +@keyframes line-dot { + 0%, + 52% { + opacity: 0; + } + 58%, + 84% { + opacity: 1; + } + 94%, + 100% { + opacity: 0; + } +} + +@media (prefers-reduced-motion: reduce) { + *, + *::before, + *::after { + animation: none !important; + transition: none !important; + scroll-behavior: auto !important; + } + .reveal { + opacity: 1; + transform: none; + } + /* With animations disabled the trend lines' base state is hidden; restore them fully drawn. */ + .anim-line-draw { + stroke-dashoffset: 0; + opacity: 1; + } + .anim-line-dot { + opacity: 1; + } +} + +/* Printing never scrolls, so scroll-reveal must not hide content on paper/PDF. */ +@media print { + .reveal { + opacity: 1; + transform: none; + } +} + +/* ---------- Minimal typography for Markdown body content (blog + CONTRACT.md) ---------- */ + +.md-body { + overflow-wrap: break-word; +} +.md-body :is(p, ul, ol, pre, blockquote, table) { + margin: 0.625rem 0; +} +.md-body li { + margin: 0.3rem 0; +} +.md-body p, +.md-body li { + line-height: 1.75; +} +.md-body :is(h1, h2, h3, h4) { + font-weight: 600; + margin: 1.25rem 0 0.5rem; +} +.md-body h1 { + font-size: 1.375rem; +} +.md-body h2 { + font-size: 1.125rem; +} +.md-body h3 { + font-size: 1rem; +} +.md-body ul { + list-style: disc; + padding-left: 1.25rem; +} +.md-body ol { + list-style: decimal; + padding-left: 1.25rem; +} +.md-body a { + @apply text-brand-700 underline decoration-brand-300 underline-offset-2 transition-colors hover:text-brand-600 dark:text-brand-300 dark:decoration-brand-700; +} +.md-body code { + @apply rounded bg-gray-100 px-1 py-0.5 text-[0.85em] text-gray-800 dark:bg-gray-800 dark:text-gray-200; +} +.md-body pre { + @apply overflow-x-auto rounded-lg border border-gray-200 bg-gray-50 p-3 text-sm dark:border-gray-800 dark:bg-gray-900; +} +.md-body pre code { + background: transparent; + color: inherit; + padding: 0; +} +.md-body blockquote { + @apply border-l-2 border-gray-300 pl-3 text-gray-600 dark:border-gray-700 dark:text-gray-400; +} +.md-body table { + border-collapse: collapse; + display: block; + overflow-x: auto; +} +.md-body :is(th, td) { + @apply border border-gray-200 px-2.5 py-1.5 text-sm dark:border-gray-800; +} +.md-body hr { + @apply my-4 border-gray-200 dark:border-gray-800; +} diff --git a/packages/landing/test/benchmark-data.test.ts b/packages/landing/test/benchmark-data.test.ts new file mode 100644 index 0000000..d10a56a --- /dev/null +++ b/packages/landing/test/benchmark-data.test.ts @@ -0,0 +1,40 @@ +import { describe, expect, it } from "vitest"; +import { + CODE_BENCH, + DATA_BENCH, + formatAccuracy, + formatPct, + formatTokensM, + formatUsd, +} from "../src/lib/benchmark-data"; + +describe("benchmark data (unified per-run means)", () => { + it("formats the data-analysis suite at its published precision", () => { + const penguin = DATA_BENCH[0]!; + expect(formatPct(penguin.accuracyPct)).toBe("66.7%"); + expect(formatAccuracy(penguin.accuracyPct, 1)).toBe("66.7"); + expect(formatTokensM(penguin.tokensM, 2)).toBe("18.04M"); + expect(formatUsd(penguin.costUsd, 3)).toBe("$0.552"); + }); + + it("formats the coding suite at its published precision (CNY converted at 7:1)", () => { + const penguin = CODE_BENCH[0]!; + expect(formatAccuracy(penguin.accuracyPct, 2)).toBe("50.00"); + expect(formatTokensM(penguin.tokensM, 2)).toBe("2.10M"); + expect(formatUsd(penguin.costUsd, 3)).toBe("$0.041"); + // 0.289 CNY / 7 -> ~0.0413 USD + expect(penguin.costUsd).toBeCloseTo(0.289 / 7, 3); + }); + + it("uses the unified framework names with PenguinHarness as the only emphasized row", () => { + for (const suite of [DATA_BENCH, CODE_BENCH]) { + expect(suite.map((r) => r.framework)).toEqual([ + "PenguinHarness", + "Claude Code", + "OpenAI Codex", + ]); + expect(suite.filter((r) => r.emphasized).map((r) => r.framework)).toEqual(["PenguinHarness"]); + for (const row of suite) expect(row.model).toBe("DeepSeek V4 Pro"); + } + }); +}); diff --git a/packages/landing/test/frontmatter.test.ts b/packages/landing/test/frontmatter.test.ts new file mode 100644 index 0000000..5203efe --- /dev/null +++ b/packages/landing/test/frontmatter.test.ts @@ -0,0 +1,18 @@ +import { describe, expect, it } from "vitest"; +import { parseFrontmatter } from "../src/lib/frontmatter"; + +describe("parseFrontmatter", () => { + it("splits meta and body, tolerating colons and quotes in values", () => { + const { meta, body } = parseFrontmatter( + `---\ntitle: "Hello: world"\ndate: 2026-07-17\ncategory: news\n---\n\n# Body\n`, + ); + expect(meta).toEqual({ title: "Hello: world", date: "2026-07-17", category: "news" }); + expect(body).toBe("# Body"); + }); + + it("returns the whole input as body when no frontmatter block exists", () => { + const { meta, body } = parseFrontmatter("# Just markdown\n"); + expect(meta).toEqual({}); + expect(body).toBe("# Just markdown"); + }); +}); diff --git a/packages/landing/test/toc.test.ts b/packages/landing/test/toc.test.ts new file mode 100644 index 0000000..33ff2ac --- /dev/null +++ b/packages/landing/test/toc.test.ts @@ -0,0 +1,26 @@ +import { describe, expect, it } from "vitest"; +import { extractToc, slugifyHeading } from "../src/lib/toc"; + +describe("blog toc", () => { + it("slugifies latin and CJK headings consistently", () => { + expect(slugifyHeading("Why PenguinHarness")).toBe("why-penguinharness"); + expect(slugifyHeading("为什么是 PenguinHarness")).toBe("为什么是-penguinharness"); + }); + + it("extracts h2/h3 headings and skips fenced code blocks", () => { + const body = [ + "intro", + "## 第一节", + "```bash", + "## not a heading", + "```", + "### 小节", + "## Second", + ].join("\n"); + expect(extractToc(body)).toEqual([ + { id: "第一节", text: "第一节", depth: 2 }, + { id: "小节", text: "小节", depth: 3 }, + { id: "second", text: "Second", depth: 2 }, + ]); + }); +}); diff --git a/packages/landing/tsconfig.json b/packages/landing/tsconfig.json new file mode 100644 index 0000000..0038d40 --- /dev/null +++ b/packages/landing/tsconfig.json @@ -0,0 +1,10 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { + "lib": ["ES2023", "DOM", "DOM.Iterable"], + "types": ["vite/client"], + "jsx": "react-jsx", + "useDefineForClassFields": true + }, + "include": ["src", "test", "vite.config.ts", "vitest.config.ts"] +} diff --git a/packages/landing/vite.config.ts b/packages/landing/vite.config.ts new file mode 100644 index 0000000..4ec2c57 --- /dev/null +++ b/packages/landing/vite.config.ts @@ -0,0 +1,17 @@ +/** + * Vite config: static landing page (React SPA + Tailwind CSS 4). + * + * BASE_PATH is injected by the GitHub Pages workflow (e.g. "/penguin-harness/") so asset + * URLs resolve under the project-pages subpath; local dev and previews default to "/". + * Blog posts are local Markdown files imported at build time via import.meta.glob (?raw), + * so the built site is fully static — no server or CMS involved. + */ +import react from "@vitejs/plugin-react"; +import tailwindcss from "@tailwindcss/vite"; +import { defineConfig } from "vite"; + +export default defineConfig({ + base: process.env.BASE_PATH ?? "/", + plugins: [react(), tailwindcss()], + server: { port: 7366 }, +}); diff --git a/packages/landing/vitest.config.ts b/packages/landing/vitest.config.ts new file mode 100644 index 0000000..a773b19 --- /dev/null +++ b/packages/landing/vitest.config.ts @@ -0,0 +1,10 @@ +/** + * Vitest config kept separate from vite.config.ts (same convention as packages/web: + * vitest's embedded vite types conflict with this package's vite 7 plugin types). + * Tests cover pure modules only, so no plugins and a node environment suffice. + */ +import { defineConfig } from "vitest/config"; + +export default defineConfig({ + test: { environment: "node" }, +}); diff --git a/packages/server/README.md b/packages/server/README.md new file mode 100644 index 0000000..5cb0cb6 --- /dev/null +++ b/packages/server/README.md @@ -0,0 +1,45 @@ +# @prismshadow/penguin-server + +The PenguinHarness Web backend — the Web implementation of the SDK's Human boundary. HTTP carries Prompt input, approvals and interrupts; Server-Sent Events stream the OmniMessage output. Adds multi-user auth, Project authorization, Session runtime, scheduling and usage accounting on top of `@prismshadow/penguin-core`. + +## Architecture + +- **HTTP**: Hono + `@hono/node-server`; `createApp(deps)` is pure assembly (no port bind — tests drive it via `app.request()`); `index.ts` is the startup entry (dotenv, graceful shutdown). +- **Storage**: SQLite via Node's built-in `node:sqlite` (WAL) holds only indexes and aggregates (users, auth sessions, Project authorization, Agent/Session indexes, usage, UI prefs, error records). Agent State, Traces and Workspaces stay as files under `~/.penguin/data//agents//`, fully shared with the SDK and CLI. +- **Runtime**: a session manager keeps active Sessions (get-or-resume-or-heal, per-Session mutex, run/compact driving); approvals surface over SSE as `approval_request` and decisions re-read the stored approval mode each time; interrupts converge pending approvals to deny before aborting; a scheduler fires `agent_state/schedule/*.toml` tasks while the service runs. +- **SSE**: per-channel monotonic event ids with a bounded replay buffer (1000 events / 2MB); reconnects replay from `Last-Event-ID` or receive `resync_required`; heartbeat comment every 20s. +- **Usage**: `token_usage` events are persisted row by row; costs are computed at query time from current per-model pricing. + +The full route tables and the SSE protocol are documented in the [Server API reference](https://prism-shadow.github.io/penguin-harness/docs/server-api). DTO types are exported for type-only import via `@prismshadow/penguin-server/api`. + +## Environment + +| Variable | Meaning | Default | +| --- | --- | --- | +| `PORT` / `HOST` | Listen port / address | `7364` / `127.0.0.1` | +| `PENGUIN_HOME` | Data root (shared with SDK/CLI) | `~/.penguin/data` | +| `PENGUIN_WEB_DB` | SQLite file path | `/web.db` | +| `PENGUIN_WEB_DIST` | Front-end build dir (static hosting + SPA fallback when present) | `../web/dist`, or the bundled `web-dist/` in the npm package | + +`.env` in the process cwd is loaded automatically. + +## Running + +```bash +pnpm --filter @prismshadow/penguin-server dev # tsx watch (front end via the Vite dev proxy) +pnpm --filter @prismshadow/penguin-server build # tsup → dist/ +pnpm --filter @prismshadow/penguin-server start # node dist/index.js +``` + +`pnpm typecheck / test` run tsc and vitest (tests use a temp root + in-memory DB; no ports, no live LLM calls). + +## Security notes (known MVP limits) + +- **CSRF**: session cookie is `SameSite=Lax` and writes accept only `Content-Type: application/json`; no CSRF token yet. +- **No login rate limiting**: add throttling at a reverse proxy for public deployments. +- **Built-in admin starts as `admin` / `admin123`**: change it immediately (a banner keeps reminding until you do). +- Passwords use `node:crypto` scrypt (`scrypt$N$r$p$salt$hash`, timingSafeEqual); login sessions renew on a 7-day sliding window; the DB stores only the token's sha256. +- Model credentials live in the Project's hidden 0600 config file; the API always masks them. +- Behind a reverse proxy, disable response buffering for SSE paths (the server already sends `X-Accel-Buffering: no`) and forward `x-forwarded-proto` to enable Secure cookies. + +Part of [PenguinHarness](https://github.com/Prism-Shadow/penguin-harness) · Apache-2.0 diff --git a/packages/server/package.json b/packages/server/package.json new file mode 100644 index 0000000..e094696 --- /dev/null +++ b/packages/server/package.json @@ -0,0 +1,59 @@ +{ + "name": "@prismshadow/penguin-server", + "version": "0.0.1", + "type": "module", + "description": "PenguinHarness Web 服务端:多用户认证与授权、Session 运行与 SSE 流式通道、用量统计,引用 @prismshadow/penguin-core。", + "license": "Apache-2.0", + "repository": { + "type": "git", + "url": "git+https://github.com/Prism-Shadow/penguin-harness.git", + "directory": "packages/server" + }, + "exports": { + ".": { + "types": "./dist/index.d.ts", + "import": "./dist/index.js" + }, + "./api": { + "types": "./dist/api/types.d.ts", + "import": "./dist/api/types.js" + } + }, + "main": "./dist/index.js", + "types": "./dist/index.d.ts", + "engines": { + "node": ">=24" + }, + "scripts": { + "dev": "pnpm --filter @prismshadow/penguin-core build && tsx watch src/index.ts", + "start": "node --disable-warning=ExperimentalWarning dist/index.js", + "typecheck": "tsc --noEmit -p tsconfig.json", + "test": "vitest run --passWithNoTests", + "build": "tsup" + }, + "dependencies": { + "@hono/node-server": "^1.15.0", + "@prismshadow/penguin-core": "workspace:*", + "@prismshadow/penguin-skills": "workspace:*", + "dotenv": "^17.0.0", + "hono": "^4.8.0", + "smol-toml": "^1.3.0", + "tar": "^7.5.20", + "yaml": "^2.5.0" + }, + "devDependencies": { + "@types/node": "^24.0.0", + "tsup": "^8.3.0", + "tsx": "^4.20.0", + "typescript": "^5.6.0", + "vitest": "^2.1.0" + }, + "files": [ + "dist", + "web-dist", + "LICENSE" + ], + "publishConfig": { + "access": "public" + } +} diff --git a/packages/server/src/api/types.ts b/packages/server/src/api/types.ts new file mode 100644 index 0000000..4b03b6e --- /dev/null +++ b/packages/server/src/api/types.ts @@ -0,0 +1,1087 @@ +/** + * Web API DTO contract — request/response types shared between server routes and the + * frontend SPA (single source of truth). + * + * These field definitions are authoritative for the Web API contract. Conventions: + * - DTO fields use camelCase; OmniMessage keeps the core protocol as-is (snake_case shell), + * no conversion; + * - This file holds only types, no implementation; exposed to the frontend via package + * exports `"./api"` for type-only import; + * - Types are taken only from core's pure subpaths (omnimessage / interfaces), so the + * frontend can safely reference them. + * + * Docs: packages/docs/content/server-api.{zh,en}.md (site path /docs/server-api) is the + * public route/SSE reference for this contract — keep it in sync when changing DTOs. + */ +import type { OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core/omnimessage"; +import type { + MCPServerConfig, + ThinkingLevelName, + ToolDefinitionConfig, +} from "@prismshadow/penguin-core/interfaces"; + +// --------------------------------------------------------------------------- +// General +// --------------------------------------------------------------------------- + +/** Unified error response body; `code` is a machine-readable error code, `message` is a Chinese user-facing message. */ +export interface ErrorBody { + error: { code: string; message: string }; +} + +/** Session approval mode (reuses the CLI enum). */ +export type ApprovalMode = "allow-all" | "deny-all" | "read-only" | "always-ask"; + +/** Session run status: idle / Task in progress / compacting. */ +export type SessionStatus = "idle" | "running" | "compacting"; + +/** Session source marker (default = user-created): triggered by Schedule / registered as a subagent session. */ +export type SessionSource = "schedule" | "subagent"; + +// --------------------------------------------------------------------------- +// Authentication and users +// --------------------------------------------------------------------------- + +export interface UserInfo { + /** Semantic id, i.e. login name: `^[a-z][a-z0-9_-]{1,31}$`, immutable after creation. */ + userId: string; + /** Built-in admin (seeded at startup). */ + isAdmin: boolean; + /** Still using the initial password (seeded/set by admin): frontend prompts the user to change it soon. */ + passwordIsInitial: boolean; + createdAt: string; +} + +export interface AuthLoginRequest { + userId: string; + password: string; +} + +export interface AuthResponse { + user: UserInfo; +} + +export interface MeResponse { + user: UserInfo; +} + +export interface PasswordChangeRequest { + oldPassword: string; + /** At least 8 characters. */ + newPassword: string; +} + +// --------------------------------------------------------------------------- +// Admin user backend (admin only) +// --------------------------------------------------------------------------- + +export interface AdminUsersResponse { + users: UserInfo[]; +} + +export interface AdminUserCreateRequest { + /** Username, i.e. user_id: `^[a-z][a-z0-9_-]{1,31}$`. */ + userId: string; + /** Initial password (at least 8 characters), flagged as an initial password. */ + password: string; +} + +export interface AdminUserCreateResponse { + user: UserInfo; +} + +export interface AdminPasswordResetRequest { + /** New initial password (at least 8 characters); resets invalidate all of the user's sessions. */ + password: string; +} + +/** User UI preferences (SQLite ui_prefs, free-form JSON; known keys declared here). */ +export interface UiPrefs { + theme?: "light" | "dark"; + lastProjectId?: string; + /** Whether the "no API key configured" guide has already been shown: once ever (on first visit to the chat page). */ + credentialGuideSeen?: boolean; + [key: string]: unknown; +} + +export interface PrefsResponse { + prefs: UiPrefs; +} + +// --------------------------------------------------------------------------- +// Project and member authorization +// --------------------------------------------------------------------------- + +export type ProjectRole = "owner" | "member"; + +export interface ProjectSummary { + projectId: string; + /** Display name (the `name` in project_config.toml); frontend falls back to projectId when unset. */ + name?: string; + /** Current user's role in this Project. */ + role: ProjectRole; + ownerUserId: string; + createdAt: string; +} + +export interface ProjectsResponse { + projects: ProjectSummary[]; +} + +export interface ProjectCreateRequest { + /** + * Semantic id, specified by the creator: `^[a-z][a-z0-9_-]{1,63}$`, immutable after creation. + * Non-admins must prefix it with `-` (the web input locks the prefix segment); + * admins are unrestricted. + */ + projectId: string; + /** Display name; defaults to projectId. */ + name?: string; +} + +export interface ProjectCreateResponse { + project: ProjectSummary; +} + +export interface MemberInfo { + userId: string; + role: ProjectRole; + createdAt: string; +} + +export interface MembersResponse { + members: MemberInfo[]; +} + +export interface MemberAddRequest { + /** Username of the user being granted access (owner invites by username). */ + userId: string; +} + +export interface MemberAddResponse { + member: MemberInfo; +} + +// --------------------------------------------------------------------------- +// Model and credential config (single .project_config.toml file; credentials are inlined on model entries) +// --------------------------------------------------------------------------- + +/** + * Model reference DTO: `(provider, modelId)` pair. + * `modelId` is the upstream request id, sent to AgentHub as-is — `/` string + * concatenation is forbidden throughout the pipeline. + */ +export interface ModelRefDto { + provider: string; + modelId: string; +} + +/** Three pricing buckets, in USD per million tokens (unit is fixed at usd_per_mtok; not carried in the DTO). */ +export interface ModelPricingDto { + cacheRead: number; + cacheWrite: number; + output: number; +} + +/** Read-only credential display: masked key and creation time; plaintext is never sent. */ +export interface CredentialInfo { + apiKeyMasked?: string; + baseUrl?: string; + createdAt?: string; +} + +export interface ModelInfo { + /** Provider group id (anthropic / openai / …, see core's MODEL_PROVIDERS; custom models use `custom`). */ + provider: string; + /** Upstream model id (the request id actually sent to AgentHub); paired with `provider` forms the entry's unique key. */ + modelId: string; + /** Display name: explicit TOML field (user-edited) takes priority, then the built-in catalog; falls back to unset (frontend shows modelId). */ + displayName?: string; + contextWindow?: number; + /** AgentHub client protocol (`openai`, etc.); defaults to AgentHub inferring it from modelId. */ + clientType?: string; + /** + * Whether image input (vision/multimodal) is supported: the TOML `vision` annotation takes + * priority, falling back to the built-in catalog annotation; if neither exists, defaults to + * unset (= treated as supported). + */ + vision?: boolean; + pricing?: ModelPricingDto; + /** Environment variable name to fall back to when api_key is empty (e.g. ANTHROPIC_API_KEY); unset if no known fallback. */ + envKey?: string; + credential?: CredentialInfo; + isDefault: boolean; +} + +export interface ModelsResponse { + /** Paired reference to the default Model. */ + defaultModel?: ModelRefDto; + /** Vision model used as a proxy reader for read_image (describes images when the session model has vision=false). */ + visionModel?: ModelRefDto; + models: ModelInfo[]; +} + +/** PUT full-table replace semantics: models not present are deleted; omitting apiKey = keep existing value. Key = (provider, modelId). */ +export interface ModelUpdateEntry { + /** Provider group (an independent entry field, always submitted with the request). */ + provider: string; + /** Upstream model id (sent to AgentHub as-is). */ + modelId: string; + /** Display name; the server does not persist it when it matches the built-in catalog (keeps the config file clean). */ + displayName?: string; + /** + * The pair reference this entry was renamed from (provided when either the group or the + * upstream id changes): the server uses this to migrate the original entry's credential + * and unknown fields to the new key — otherwise a full-table replace would delete the + * original entry along with its credential. + */ + renamedFrom?: ModelRefDto; + contextWindow?: number; + /** Empty string/omitted = unspecified (AgentHub infers it from modelId). */ + clientType?: string; + /** Whether image input (vision/multimodal) is supported; omitted = supported (not persisted). */ + vision?: boolean; + pricing?: ModelPricingDto; + /** Providing it overwrites and updates createdAt; omitting it keeps the existing value. */ + apiKey?: string; + /** When true, clears the stored api_key. */ + clearApiKey?: boolean; + /** null clears it; omitted keeps the existing value. */ + baseUrl?: string | null; +} + +export interface ModelsUpdateRequest { + /** Must be included in models (matched by paired reference). */ + defaultModel?: ModelRefDto; + /** Vision model used as a proxy reader for read_image: must be included in models and not annotated vision=false; omitted keeps the existing value. */ + visionModel?: ModelRefDto; + models: ModelUpdateEntry[]; +} + +/** + * Connectivity test (POST /api/projects/:p/models/test): the model reference is submitted as + * a pair in the request body; the rest are optional overrides (for trying out an unsaved + * config). When the model isn't in the config yet (adding a custom model — test-before-save), + * all parameters come from this request body. + */ +export interface ModelTestRequest { + /** Provider group of the model under test (paired with modelId). */ + provider: string; + /** Upstream id of the model under test (sent to AgentHub as-is). */ + modelId: string; + /** Newly entered API key (plaintext); used for the test if provided. */ + apiKey?: string; + /** "Clear saved API key" is checked: the test does **not** fall back to the stored key (tests against the current draft). */ + clearApiKey?: boolean; + /** + * base URL (not secret; the frontend always sends the form's current value): a string + * means use it, `null` means explicitly clear it (no fallback to the stored value), + * `undefined` means fall back to the stored value only when not provided. + */ + baseUrl?: string | null; + /** AgentHub client protocol; required for unsaved custom models (otherwise the id can't be auto-routed). */ + clientType?: string; +} + +/** Connectivity test result: carries round-trip latency when ok, and a reason on failure (truncated raw provider error). */ +export interface ModelTestResponse { + ok: boolean; + latencyMs?: number; + message?: string; +} + +// --------------------------------------------------------------------------- +// Vault environment variables (Agent-level: agent_state/.vault.toml) +// --------------------------------------------------------------------------- + +/** Read-only vault entry display: key name + masked value; plaintext is never sent. */ +export interface VaultEntryInfo { + key: string; + valueMasked: string; +} + +export interface VaultResponse { + entries: VaultEntryInfo[]; +} + +/** A single entry under PUT full-table replace semantics: omitting value = keep the existing value (required for new keys). */ +export interface VaultEntryUpdate { + /** Shell environment variable name rule: starts with a letter or underscore, followed by letters/digits/underscores only. */ + key: string; + /** Non-empty string; omitted keeps the existing value. */ + value?: string; +} + +/** PUT full-table replace semantics (same as models): keys not present in the body are deleted. */ +export interface VaultUpdateRequest { + entries: VaultEntryUpdate[]; +} + +// --------------------------------------------------------------------------- +// Agent and its config (system_config.yaml + AGENTS.md) +// --------------------------------------------------------------------------- + +export interface AgentSummary { + agentId: string; + name?: string; + description?: string; + createdAt?: string; + /** Last config modification time: the larger mtime of system_config.yaml / AGENTS.md (unset if stat fails). */ + updatedAt?: string; + /** Number of this Agent's Sessions currently running / compacting. */ + activeSessionCount: number; + /** Total Session count (DB index ∪ Trace directory discovery, including archived). */ + sessionCount: number; + /** Daily active Session count for the last 30 days (index 0 = earliest, last = today; active = created that day or has a Trace record that day). */ + sessionActivity: number[]; + /** Tool count: number of tools.builtin + tools.mcpServers config entries (MCP counted per server). */ + toolCount: number; + /** Agent State version number (the `version` in system_config.yaml; treated as 1 if missing). */ + version: number; + /** Vault key count (number of keys in agent_state/.vault.toml). */ + vaultKeyCount: number; + /** Schedule count (number of .toml files under agent_state/schedule/, including invalid ones). */ + scheduleCount: number; +} + +export interface AgentsResponse { + agents: AgentSummary[]; +} + +export interface AgentCreateRequest { + /** Semantic id, specified by the creator: `^[a-z][a-z0-9_-]{1,63}$`, unique within the Project, immutable after creation. */ + agentId: string; + /** Display name; defaults to agentId. */ + name?: string; + description?: string; +} + +export interface AgentCreateResponse { + agent: AgentSummary; +} + +export interface AgentModelConfigDto { + maxTokens?: number; + thinkingLevel?: ThinkingLevelName; + timeoutMs?: number; +} + +export interface AgentCompactionConfigDto { + maxContextLength?: number; + maxSessionTurns?: number; + mode?: "summarize" | "discard"; + prompt?: string; +} + +/** Structured view of system_config.yaml (for the edit form). */ +export interface AgentConfigDto { + name?: string; + description?: string; + /** Agent State version number (treated as 1 if missing; shown in the settings page overview). */ + version: number; + systemPrompt: string; + maxTurns?: number; + model?: AgentModelConfigDto; + compaction?: AgentCompactionConfigDto; + toolsBuiltin: ToolDefinitionConfig[]; + mcpServers: MCPServerConfig[]; +} + +export interface AgentConfigResponse { + agentsMd: string; + /** Raw system_config.yaml text (read-only display / diagnostics). */ + systemConfigYaml: string; + config: AgentConfigDto; + /** Agent State absolute path. */ + stateDir: string; + activeSessionCount: number; +} + +/** PUT any subset: only provided keys are updated (remaining YAML content and comments preserved); agentsMd overwrites the whole file. */ +export interface AgentConfigUpdateRequest { + agentsMd?: string; + config?: { + name?: string; + description?: string; + systemPrompt?: string; + maxTurns?: number; + model?: AgentModelConfigDto; + compaction?: AgentCompactionConfigDto; + toolsBuiltin?: ToolDefinitionConfig[]; + mcpServers?: MCPServerConfig[]; + }; +} + +// --------------------------------------------------------------------------- +// Session +// --------------------------------------------------------------------------- + +export interface SessionInfo { + sessionId: string; + projectId: string; + agentId: string; + /** Provider group of the session's model (paired with `modelId` to form a model reference). */ + provider: string; + /** Upstream model_id of the session's model (the request id sent to AgentHub). */ + modelId: string; + workspace: string; + approvalMode: ApprovalMode; + /** Short title auto-generated by the model after the first turn; unset until generated (frontend shows "New Chat"). */ + title?: string; + /** Session source (for list badges); unset for user-created sessions. */ + source?: SessionSource; + createdAt: string; + status: SessionStatus; + /** Number of approvals awaiting human decision (a persisted count outside server events, for list badges). */ + pendingApprovalCount: number; + /** Whether a Trace record exists (a Task has been started). */ + hasTrace: boolean; + /** Whether archived (hidden from the default list, grouped under "Archived"). */ + archived: boolean; +} + +export interface SessionsResponse { + sessions: SessionInfo[]; +} + +/** Server directory browsing (advanced new-Workspace picker): starts from the home directory by default, can navigate up to the root. */ +export interface DirEntryInfo { + name: string; + /** Absolute path of this subdirectory (can be submitted directly as a Workspace). */ + path: string; +} +export interface DirListResponse { + /** Absolute path of the current directory (realpath). */ + path: string; + /** Absolute path of the parent directory; null when already at the root. */ + parent: string | null; + /** Subdirectory list (sorted by name, files excluded). */ + entries: DirEntryInfo[]; +} + +export interface SessionCreateRequest { + /** Upstream id of the session's model (paired with provider); defaults to the Project's default Model. */ + modelId?: string; + /** + * Provider group for `modelId`; when omitted, resolved via resolveModelRef semantics — + * modelId can only be resolved if it's globally unique by exact match in the config; + * 0 or multiple matches return 400. + */ + provider?: string; + /** Any existing directory on the server; defaults to auto-creating a temporary Workspace. */ + workspace?: string; + /** Defaults to allow-all. */ + approvalMode?: ApprovalMode; +} + +export interface SessionCreateResponse { + session: SessionInfo; +} + +export interface SessionResponse { + session: SessionInfo; +} + +export interface SessionPatchRequest { + approvalMode?: ApprovalMode; + /** Archive / unarchive (default list hides archived). */ + archived?: boolean; + /** Manual rename; non-empty string, overrides the auto-generated title. */ + title?: string; +} + +/** Message history: the full messages and events from concatenating all of this Session's Trace files in order (excludes partial_*). */ +export interface MessagesResponse { + messages: OmniMessage[]; +} + +// --------------------------------------------------------------------------- +// Task run, approval, interruption, compaction +// --------------------------------------------------------------------------- + +/** + * A single Prompt's input parts: text or image (data: / http(s) URL). + * Docs: /docs/server-api § "Session-Level Endpoints". + */ +export type TaskInputPart = + { type: "text"; text: string } | { type: "image_url"; imageUrl: string }; + +export interface TaskCreateRequest { + input: TaskInputPart[]; +} + +export interface TaskCreateResponse { + /** Current actual session_id: a Trace-less invalid Session self-heals and returns a new id; the frontend updates its route accordingly. */ + sessionId: string; +} + +export interface ApprovalDecisionRequest { + decision: "allow" | "deny"; +} + +// --------------------------------------------------------------------------- +// SSE server events (OmniMessage uses the default event, only server_event here) +// --------------------------------------------------------------------------- + +/** Docs: /docs/server-api § "Streaming (SSE)". */ +export type ServerEvent = + /** + * Approval request escalated to a human: every call under always-ask, plus rw/unknown-permission + * calls under read-only (see runtime/approvals.ts); pending approvals are resent on reconnect. + */ + | { type: "approval_request"; toolCall: OmniMessage; origin?: string[] } + /** Session run status flip (for toggling the input area and list). */ + | { type: "task_state"; state: SessionStatus } + /** The model-generated title after the first turn has been persisted (for in-place list updates). */ + | { type: "session_title"; sessionId: string; title: string } + /** Last-Event-ID has been evicted from the buffer: the frontend should re-fetch the history endpoint before continuing to consume this connection. */ + | { type: "resync_required" } + /** Placeholder handshake on the user channel (reserved for automated task notifications). */ + | { type: "hello" } + /** New session registered (pushed over the parent session's channel for subagent sessions): frontend refreshes the list in place. */ + | { + type: "session_created"; + projectId: string; + agentId: string; + sessionId: string; + source: SessionSource; + } + | ScheduleServerEvent; + +/** Schedule notification (user-level event stream; firing and delivery are notified via /api/events). */ +export type ScheduleServerEvent = + /** Fired and sent (sessionId is the session that received the Prompt; a new session under new-Session mode). */ + | { type: "schedule_fired"; projectId: string; agentId: string; name: string; sessionId: string } + /** Target Session is running; this firing is queued and will be sent once it's idle. */ + | { + type: "schedule_queued"; + projectId: string; + agentId: string; + name: string; + sessionId: string; + }; + +// --------------------------------------------------------------------------- +// Trace browsing and performance analysis +// --------------------------------------------------------------------------- + +export interface TraceFileInfo { + /** Trace file index (one file corresponds to one complete model context). */ + index: number; + /** Date subdirectory it belongs to (yyyy-mm-dd). */ + date: string; + sizeBytes: number; + mtime: string; +} + +export interface SessionTracesResponse { + files: TraceFileInfo[]; +} + +export interface TraceEventsResponse { + events: OmniMessage[]; + offset: number; + limit: number; + /** Total line count of the file (basis for pagination). */ + total: number; +} + +/** Duration span of a single LLM Request (request_begin/request_end paired by proximity). */ +export interface RequestSpan { + beginTs: string; + endTs?: string; + durationMs?: number; + status?: string; + /** The Task it belongs to (same convention as modelSegments/toolSpans). */ + taskIndex: number; + /** Compaction request (falls between compaction_begin and compaction_end): excluded from TPS, see TraceTaskStats. */ + compaction?: boolean; + /** + * Total human approval wait time within this Request. core does `await approve(tc)` inside + * the streaming loop — if approval doesn't return, the next chunk isn't consumed and + * `request_end` can't be emitted either, so the entire human wait falls inside the span + * (see context-engine's runTurn). Tool **execution** is not included (`void executeOne`, + * doesn't block the loop). + */ + approvalWaitMs?: number; + /** LLM generation duration = durationMs − approvalWaitMs (≥ 0): only this can be used as the TPS denominator, not durationMs. */ + activeMs?: number; +} + +/** + * Per-Task Token / duration figures (aggregated server-side over the **entire** Trace file, + * aligned with the Chat page's task-stats). + * + * Provided separately instead of letting the frontend aggregate `requests` + events itself: + * the frontend's events are paginated (only the first N), so self-aggregation would mismatch + * a numerator covering only the first N against a denominator covering the whole file. + */ +export interface TraceTaskStats { + taskIndex: number; + /** + * This turn is a **compaction turn** (compaction forms its own turn); the UI marks it with + * a "Compaction" badge accordingly. It's treated the same as a user turn: it has Token / + * cost / duration / TPS, and **counts normally toward global stats** — the global totals are + * just the sum of the per-turn cards below, the two scopes match, so adding up the per-turn + * numbers must equal the total. + */ + compaction?: boolean; + /** + * This turn's message index range within the **entire file** (inclusive). A single + * sequential scan on the server tells which turn each message belongs to; the frontend + * attributes messages by this, **no longer guessing by timestamp** — the same millisecond + * can pack "previous turn's last reply + compaction start + compaction prompt + next turn's + * request_begin", which time boundaries can't separate, misattributing this turn's reply to + * the next turn. + */ + messageFrom: number; + messageTo: number; + /** + * This turn's duration span: `startTs` = the moment of this turn's **first `request_begin`** + * — duration only looks at LLM requests, not the timestamp of user text like the user + * Prompt / compaction summary (`` is created during compaction but only + * persisted on the next run; resuming the next day would inflate the first turn by a whole + * day for no reason); `endTs` = the moment of the last non-session_meta message in the + * range. For a degenerate turn with no Request at all (interrupted right after sending), + * `startTs` is an empty string and duration counts as 0. + */ + startTs: string; + endTs: string; + /** + * Context usage at the end of this Task = the three-bucket Token snapshot of the last + * **non-compaction** Request (same convention as the Chat page's `contextNow`). Note this + * must not be the sum of this Task's Requests — each Request's input carries the full + * history again, so summing double-counts the context, and a few rounds of tool calls + * would blow past the context window. A pure-compaction Task (no non-compaction Request) + * has no value here. + */ + context?: { cacheRead: number; cacheWrite: number; output: number }; + /** + * This turn's **cumulative** usage (the sum of the three buckets over every Request in this + * Task), for Token stats and cost conversion. Two different figures from `context`: that one + * is a snapshot (how much is occupied right now), this one is a ledger (how much this turn + * spent in total). Includes compaction requests — compaction tokens are real money spent and + * must be counted; consistent with the Chat page's tokensByBucket. + */ + tokens: { cacheRead: number; cacheWrite: number; output: number }; + /** + * Total LLM generation duration for this turn (the denominator for output TPS; human + * approval wait already deducted). The numerator is simply `tokens.output`: since + * compaction forms its own turn, each turn's output tokens are just its own Requests' + * output — there's no second figure to reconcile. + */ + llmMs: number; +} + +/** Duration span of a single tool call (complete tool_call message → paired tool_call_output). */ +export interface ToolCallSpan { + toolCallId: string; + name: string; + startTs: string; + endTs?: string; + durationMs?: number; + stopReason?: string; +} + +/** Workspace file entry (Files tab). */ +export interface WorkspaceFileEntry { + name: string; + kind: "dir" | "file"; + sizeBytes: number; + mtime: string; +} + +export interface WorkspaceFilesResponse { + /** Requested relative path ("" = Workspace root). */ + path: string; + entries: WorkspaceFileEntry[]; +} + +/** Batch file existence check (message file cards only list files that actually exist). */ +export interface FilesStatRequest { + /** Paths relative to the Workspace root (≤100 items, each ≤512 characters). */ + paths: string[]; +} + +export interface FilesStatResponse { + /** Confirmed existing paths (regular files within bounds), preserving request order and deduplicated; out-of-bounds and resolution failures count as non-existent. */ + existing: string[]; +} + +/** + * Model serial segments (autoregressive decoding): Trace records completion times, so each + * segment's duration = its own time − the previous event's time (the request's first segment + * is based on request_begin; user input is treated as sent instantaneously and takes no + * segment). + */ +export interface TraceModelSegment { + kind: "thinking" | "text" | "tool_call"; + startTs: string; + endTs: string; + /** Given when kind=tool_call. */ + toolCallId?: string; + name?: string; + /** The Task it belongs to (a single user turn can contain multiple Requests): the frontend groups by this, each Task on its own independent timeline. */ + taskIndex: number; +} + +/** + * Tool full lifecycle (parallel to model decoding): initiated (callTs) → approved + * (approvalTs) → output (outputTs). Unclosed fields are unset (approval pending / executing / + * file truncated). + */ +export interface TraceToolSpan { + toolCallId: string; + name: string; + callTs: string; + approvalTs?: string; + decision?: string; + outputTs?: string; + stopReason?: string; + /** The Task that initiated this tool (grouped with its tool_call segment): async output belongs to this Task even if it arrives after request_end. */ + taskIndex: number; +} + +export interface UsageTrendPointInTrace { + ts: string; + requestTotal: number; + sessionTotal: number; +} + +export interface TraceAnalysisResponse { + /** + * Sum of all turns' durations (**including compaction turns**, same scope as `tasks` — the + * global figure is just the sum of the per-turn figures below; gaps between turns where the + * user is thinking or away are not counted). Computed server-side over the entire file: the + * frontend's events are paginated, so self-aggregation would undercount. + */ + elapsedMs: number; + requests: RequestSpan[]; + /** Token / duration aggregated per Task (used directly by the Trace page's context ring and per-turn TPS). */ + tasks: TraceTaskStats[]; + toolCalls: ToolCallSpan[]; + /** Execution timeline: model serial segments (LLM lane). */ + modelSegments: TraceModelSegment[]; + /** Execution timeline: each tool's approval/execution phases (independent lane, can overlap with model decoding). */ + toolSpans: TraceToolSpan[]; + /** Number of request_end events with status ∈ {timeout, malformed}. */ + reconnectCount: number; + /** Number of compaction_begin events. */ + compactionCount: number; + usageTrend: UsageTrendPointInTrace[]; +} + +export interface AgentTraceFileRef { + index: number; + sizeBytes: number; +} + +export interface AgentTraceSessionGroup { + sessionId: string; + files: AgentTraceFileRef[]; +} + +export interface AgentTraceDateGroup { + date: string; + sessions: AgentTraceSessionGroup[]; +} + +/** Agent → date → Session → Trace file drill-down browsing structure (reverse chronological). */ +export interface AgentTracesResponse { + dates: AgentTraceDateGroup[]; +} + +// --------------------------------------------------------------------------- +// Usage and cost statistics +// --------------------------------------------------------------------------- + +export type UsageGroupBy = "date" | "agent" | "model" | "session"; + +export interface UsageBucket { + total: number; + requests: number; + /** Cost converted using current pricing at query time (USD); a partial sum when uncosted Models are included, null if none has pricing. */ + cost: number | null; + /** Whether any Model has no pricing (its usage isn't included in cost; counted once pricing is added later). */ + hasUncosted: boolean; +} + +export interface UsageGroupRow { + /** Group key: date / agentId / modelId / sessionId. */ + key: string; + /** Provider group when groupBy=model (rows are broken down by (provider, modelId); unset for other dimensions). */ + provider?: string; + cacheRead: number; + cacheWrite: number; + output: number; + total: number; + requests: number; + cost: number | null; + hasUncosted: boolean; +} + +export interface UsageTrendPoint { + date: string; + total: number; + cost: number | null; + /** Daily Token buckets (for the cost center's "Token Changes" stacked chart: cacheRead/cacheWrite/output). */ + cacheRead: number; + cacheWrite: number; + output: number; +} + +/** Invocation count per Agent (for the cost center's "Agent Invocation Count" chart). */ +export interface UsageAgentCount { + agentId: string; + requests: number; + total: number; +} + +/** Request success rate per Model (for the cost center's "Model Success Rate" chart; rows broken down by (provider, modelId)). */ +export interface UsageSuccessRate { + provider: string; + modelId: string; + /** Number of successful requests. */ + completed: number; + /** Success rate denominator = all requests − aborted (user-initiated interruption isn't a model failure and shouldn't lower the success rate). */ + total: number; + /** Count of user interruptions (excluded from success rate, shown separately). */ + aborted: number; + /** Failure breakdown (shown on hover; unknown statuses count toward total but not these three). */ + failed: number; + timeout: number; + malformed: number; +} + +/** Occurrence count of an error for a given source · code (the "most common" metric in the stats center's error panel). */ +export interface UsageErrorCount { + source: string; + code: string; + kind: string; + count: number; +} + +/** A single error summary (one row in the stats center's error panel table). */ +export interface UsageErrorItem { + ts: string; + source: string; + code: string; + kind: string; + message: string; +} + +/** + * Server-side error capture stats: not affected by the model + * filter (HTTP / process errors have no Model dimension), but affected by date and agent + * filters. Errors with no Project attribution (login, process-level) are counted in every + * Project's view. The stats center presents this as "summary stats + detail table" with no + * chart, so it only has a total count, the most common error code, and the most recent N + * items. + */ +export interface UsageErrors { + total: number; + /** Count of unexpected ones (500 / runtime exceptions) among them — the part the frontend highlights. */ + unexpected: number; + /** The most frequent source · code (null when there are no errors). */ + topCode: UsageErrorCount | null; + /** Most recent N items (reverse chronological). */ + recent: UsageErrorItem[]; +} + +export interface UsageResponse { + summary: { + today: UsageBucket; + last7d: UsageBucket; + total: UsageBucket; + }; + groupBy: UsageGroupBy; + groups: UsageGroupRow[]; + /** Daily trend for the last 30 days (includes Token buckets and cost; affected by agent/model filters). */ + trend: UsageTrendPoint[]; + /** Invocation count per Agent (affected by date/model filters). */ + byAgent: UsageAgentCount[]; + /** Raw success rate counts per Model (affected by date/agent filters). */ + success: UsageSuccessRate[]; + /** Server-side error capture stats (affected by date/agent filters; unaffected by model filter). */ + errors: UsageErrors; + /** List of Agent ids that have appeared in this Project (for the filter dropdown; unaffected by current filters). */ + agentIds: string[]; + /** List of Model paired references that have appeared in this Project (for the filter dropdown). */ + models: ModelRefDto[]; +} + +// --------------------------------------------------------------------------- +// Schedule +// --------------------------------------------------------------------------- + +/** Display-facing schedule status: the file's `enabled` only expresses intent; the rest is derived from runtime state. */ +export type ScheduleStatus = "active" | "disabled" | "expired" | "done" | "missed" | "invalid"; + +export interface ScheduleItem { + /** Filename (without .toml) is the identifier. */ + name: string; + prompt: string; + enabled: boolean; + /** ISO 8601. */ + startAt: string; + /** Raw fixed interval (e.g. `30m`); unset means a one-off task. */ + period?: string; + endAt?: string; + /** Bound target Session; defaults to creating a new Session each time. */ + sessionId?: string; + workspace?: string; + /** Model for new-Session mode (upstream id, paired with provider); defaults to the Project's default reference. */ + modelId?: string; + /** Provider group for `modelId`; when omitted, resolved via resolveModelRef semantics (resolvable only on a unique match). */ + provider?: string; + status: ScheduleStatus; + invalidReason?: string; + /** Next scheduled fire time (ISO 8601); unset when done/missed/invalid/disabled. */ + nextFireAt?: string; + /** Most recent actual fire time (ISO 8601). */ + lastFiredAt?: string; + /** Queued, waiting for the target Session to become idle. */ + queued: boolean; + creatorUserId?: string; +} + +export interface SchedulesResponse { + schedules: ScheduleItem[]; + /** Files that failed to parse (skipped from scheduling and logged as errors). */ + invalidFiles: Array<{ name: string; error: string }>; +} + +export interface ScheduleUpsertRequest { + prompt: string; + enabled: boolean; + startAt: string; + period?: string; + endAt?: string; + sessionId?: string; + workspace?: string; + /** Model for new-Session mode (upstream id); defaults to the Project's default reference. */ + modelId?: string; + /** Provider group for `modelId`; when omitted, validated as uniquely resolvable via resolveModelRef semantics at save/reconciliation time. */ + provider?: string; +} + +// --------------------------------------------------------------------------- +// Agent State version and snapshots +// --------------------------------------------------------------------------- + +export interface AgentImportRequest { + /** Base64 of the snapshot package (tar.gz). */ + dataBase64: string; + /** Explicit confirmation is required when the package version is equal to or lower than the current version, otherwise 409. */ + confirm?: boolean; +} + +export interface AgentImportResponse { + /** Agent State version number after import (taken from the package's value). */ + version: number; +} + +// --------------------------------------------------------------------------- +// Benchmark scoring (read-only display) +// --------------------------------------------------------------------------- + +/** Raw result of a single run (a scoreboard per-case runs[] entry). */ +export interface BenchmarkRunScore { + score: number; + cost?: number; + durationMs?: number; + /** Id of the Session under test in this run (links to Trace). */ + sessionId?: string; +} + +export interface BenchmarkCaseScore { + case: string; + /** Per-case score = average of runs (equals that single run's score under the legacy single-run format). */ + score: number; + cost?: number; + durationMs?: number; + /** For legacy format compatibility: per-case single Session id (new format keeps it inside runs[]). */ + sessionId?: string; + /** Raw results per run; unset under the legacy format (the server backfills one entry when parsing as a single run). */ + runs?: BenchmarkRunScore[]; +} + +export interface BenchmarkEvaluation { + /** Evaluation timestamp (ISO 8601). */ + time: string; + /** Evaluation summary title (a one-line conclusion; shown separately from the body summary; required when generating, tolerated as unset when displaying). */ + summaryTitle?: string; + /** Evaluation summary body: how the score was derived, what optimizations were made to the Agent this round (required when generating, tolerated as unset when displaying). */ + summary?: string; + /** Model actually used for this evaluation round (upstream id, paired with provider; the chart series is split by model). */ + modelId?: string; + /** Provider group for `modelId`. */ + provider?: string; + /** Agent State version number under test. */ + version?: number; + /** Total score (sum of per-case scores; max score defined by the scoring rubric). */ + score: number; + cost?: number; + durationMs?: number; + cases: BenchmarkCaseScore[]; +} + +export interface BenchmarkSummary { + /** Directory name is the identifier (semantic naming, e.g. swe-bench-v1). */ + id: string; + /** Title from benchmark_config.toml; falls back to the directory name if unset. */ + title: string; + description?: string; + /** Number of runs per case (the `runs` field in benchmark_config.toml, ≥1; defaults to 1). */ + runs?: number; + /** Case count (number of case subfolders). */ + caseCount: number; + /** Time-ordered evaluation records (the evaluations[] in scoreboard.yaml). */ + evaluations: BenchmarkEvaluation[]; +} + +export interface BenchmarksResponse { + benchmarks: BenchmarkSummary[]; +} + +// --------------------------------------------------------------------------- +// Skill library and Agent's installed Skills +// --------------------------------------------------------------------------- + +export interface SkillMetadataItem { + /** Skill directory name (the identity key for install / uninstall / Prompt addressing). */ + name: string; + description: string; + /** Short description for frontend display (frontmatter short_description, optional; falls back to description if missing). */ + shortDescription?: string; + shortDescriptionZh?: string; + /** Custom icon (raw icon.svg text from the skill directory, optional; frontend falls back to a default book icon if missing). */ + icon?: string; + /** Version number (natural number, frontmatter version; falls back to 1 if invalid). */ + version: number; + /** Update date (YYYY-MM-DD, frontmatter updated; defaults to an empty string). */ + updated: string; +} + +export interface SkillGroupItem { + id: string; + title: string; + /** Chinese group title (optional; the UI displays it per language). */ + titleZh?: string; + skills: SkillMetadataItem[]; +} + +/** GET /api/skills: library groups and metadata (excludes body content). */ +export interface SkillLibraryResponse { + groups: SkillGroupItem[]; +} + +/** GET|POST /api/projects/:p/agents/:a/skills: Skills installed on this Agent. */ +export interface AgentSkillsResponse { + skills: SkillMetadataItem[]; +} + +/** POST install request: all names must exist in the library; already-installed ones are overwritten with library content (i.e. updated). */ +export interface SkillInstallRequest { + names: string[]; +} diff --git a/packages/server/src/app.ts b/packages/server/src/app.ts new file mode 100644 index 0000000..0e6a45e --- /dev/null +++ b/packages/server/src/app.ts @@ -0,0 +1,395 @@ +/** + * Hono app assembly: middleware + route mounting + static hosting + + * error handling. + * + * `createApp(deps)` is pure assembly (does not listen on a port): tests inject requests via + * `app.request()`; `buildAppDeps(config)` assembles all services from config (test doubles + * like SessionLoader can be injected). The startup entry point is in index.ts. + */ +import fs from "node:fs"; +import fsp from "node:fs/promises"; +import path from "node:path"; +import { Hono } from "hono"; +import type { Context } from "hono"; +import type { DatabaseSync } from "node:sqlite"; +import type { ServerConfig } from "./config.js"; +import { openDatabase } from "./db/database.js"; +import { AgentsRepo } from "./db/repos/agents.js"; +import { AuthSessionsRepo } from "./db/repos/auth-sessions.js"; +import { ErrorsRepo } from "./db/repos/errors.js"; +import { MembersRepo } from "./db/repos/members.js"; +import { ProjectsRepo } from "./db/repos/projects.js"; +import { SchedulesRepo } from "./db/repos/schedules.js"; +import { SessionsRepo } from "./db/repos/sessions.js"; +import { UiPrefsRepo } from "./db/repos/ui-prefs.js"; +import { UsageRepo } from "./db/repos/usage.js"; +import { UsersRepo } from "./db/repos/users.js"; +import type { UserRow } from "./db/repos/users.js"; +import { authMiddleware, jsonOnlyWrites } from "./auth/middleware.js"; +import type { AppEnv } from "./auth/middleware.js"; +import { AuthService } from "./auth/service.js"; +import { handleError, HttpError, errorBody } from "./http/errors.js"; +import { adminUsersRoutes } from "./http/routes/admin.js"; +import { authRoutes } from "./http/routes/auth.js"; +import { meRoutes } from "./http/routes/me.js"; +import { eventsRoutes, userChannelKey } from "./http/routes/events.js"; +import { projectsRoutes } from "./http/routes/projects.js"; +import { membersRoutes } from "./http/routes/members.js"; +import { modelsRoutes } from "./http/routes/models.js"; +import { vaultRoutes } from "./http/routes/vault.js"; +import { scheduleRoutes } from "./http/routes/schedules.js"; +import { benchmarksRoutes } from "./http/routes/benchmarks.js"; +import { agentSkillsRoutes, skillLibraryRoutes } from "./http/routes/skills.js"; +import { agentTransferRoutes } from "./http/routes/agent-transfer.js"; +import { agentsRoutes } from "./http/routes/agents.js"; +import { dirsRoutes } from "./http/routes/dirs.js"; +import { agentConfigRoutes } from "./http/routes/agent-config.js"; +import { agentTracesRoutes } from "./http/routes/agent-traces.js"; +import { usageRoutes } from "./http/routes/usage.js"; +import { agentSessionsRoutes, sessionsRoutes } from "./http/routes/sessions.js"; +import { ChannelHub } from "./runtime/channel.js"; +import { ErrorRecorder } from "./runtime/error-recorder.js"; +import { createCoreSessionLoader, SessionManager } from "./runtime/session-manager.js"; +import type { SessionLoader } from "./runtime/session-manager.js"; +import { Scheduler } from "./runtime/scheduler.js"; +import { TitleGenerator } from "./runtime/title-generator.js"; +import type { TitleNotifier } from "./runtime/title-generator.js"; +import { UsageRecorder } from "./runtime/usage-recorder.js"; +import { AdminService } from "./services/admin-service.js"; +import { AgentConfigService } from "./services/agent-config-service.js"; +import { AgentService } from "./services/agent-service.js"; +import { BenchmarkService } from "./services/benchmark-service.js"; +import { SnapshotService } from "./services/snapshot-service.js"; +import { ProjectConfigService } from "./services/project-config-service.js"; +import { ProjectService } from "./services/project-service.js"; +import { SessionService } from "./services/session-service.js"; +import { TraceService } from "./services/trace-service.js"; +import { UsageService } from "./services/usage-service.js"; +import { WorkspaceFilesService } from "./services/workspace-files-service.js"; + +/** Request body size limit (tasks may carry data: images): 20MB. */ +const MAX_BODY_BYTES = 20 * 1024 * 1024; + +export interface AppDeps { + config: ServerConfig; + db: DatabaseSync; + sessionsRepo: SessionsRepo; + prefsRepo: UiPrefsRepo; + authService: AuthService; + adminService: AdminService; + projectService: ProjectService; + projectConfigService: ProjectConfigService; + agentService: AgentService; + agentConfigService: AgentConfigService; + sessionService: SessionService; + traceService: TraceService; + usageService: UsageService; + workspaceFiles: WorkspaceFilesService; + benchmarks: BenchmarkService; + snapshots: SnapshotService; + schedulesRepo: SchedulesRepo; + scheduler: Scheduler; + channels: ChannelHub; + manager: SessionManager; + /** Error persistence (shared by app.onError and various background capture points; the process-level fallback is in index.ts). */ + errors: ErrorRecorder; + /** Request log output (minimal one-liner); tests inject a noop. */ + log: (line: string) => void; +} + +export interface BuildDepsOverrides { + /** Test double: session-manager's underlying loader (avoids the real LLM/SDK path). */ + loader?: SessionLoader; + /** Test double: Session title generator (avoids real LLM requests). */ + titles?: TitleNotifier; + log?: (line: string) => void; + now?: () => Date; +} + +/** Assemble all services from config (shared by production and tests; tests pass dbPath=":memory:" and a temp root). */ +export function buildAppDeps(config: ServerConfig, overrides: BuildDepsOverrides = {}): AppDeps { + const db = openDatabase(config.dbPath); + const log = overrides.log ?? ((line: string) => console.log(line)); + + const usersRepo = new UsersRepo(db); + const authSessionsRepo = new AuthSessionsRepo(db); + const projectsRepo = new ProjectsRepo(db); + const membersRepo = new MembersRepo(db); + const agentsRepo = new AgentsRepo(db); + const sessionsRepo = new SessionsRepo(db); + const usageRepo = new UsageRepo(db); + const errorsRepo = new ErrorsRepo(db); + const prefsRepo = new UiPrefsRepo(db); + const schedulesRepo = new SchedulesRepo(db); + + const projectConfigService = new ProjectConfigService(config.root); + const agentConfigService = new AgentConfigService(config.root); + const agentService = new AgentService(config.root, agentsRepo, agentConfigService); + const traceService = new TraceService(config.root); + const workspaceFiles = new WorkspaceFilesService(); + const benchmarks = new BenchmarkService(config.root); + const snapshots = new SnapshotService(config.root); + const usageService = new UsageService( + usageRepo, + errorsRepo, + (projectId, provider, modelId) => projectConfigService.getPricing(projectId, provider, modelId), + overrides.now ?? (() => new Date()), + ); + + // Channel idle reclamation skips active Sessions (running/compacting can go a long time + // without a publish, e.g. while waiting for approval). + // manager is created after channels: use a lazy predicate (managerRef is assigned by the + // time the sweep timer fires). + let managerRef: SessionManager | undefined; + const channels = new ChannelHub({ + isActive: (key) => managerRef !== undefined && managerRef.statusOf(key) !== "idle", + }); + const recorder = new UsageRecorder(usageRepo, overrides.now ?? (() => new Date())); + const errors = new ErrorRecorder(errorsRepo, overrides.now ?? (() => new Date())); + const titles = + overrides.titles ?? + new TitleGenerator({ sessions: sessionsRepo, channels, recorder, errors, log }); + const manager = new SessionManager({ + sessions: sessionsRepo, + channels, + loader: overrides.loader ?? createCoreSessionLoader(config.root), + recorder, + errors, + titles, + log, + }); + managerRef = manager; + + const projectService = new ProjectService({ + root: config.root, + users: usersRepo, + projects: projectsRepo, + members: membersRepo, + agents: agentsRepo, + sessions: sessionsRepo, + usage: usageRepo, + errors: errorsRepo, + schedules: schedulesRepo, + projectConfig: projectConfigService, + manager, + }); + const authService = new AuthService({ + users: usersRepo, + authSessions: authSessionsRepo, + provisionInitialProject: (user, isAdmin) => + projectService.provisionInitialProject(user, isAdmin), + sessionTtlMs: config.authSessionTtlMs, + sessionRenewMs: config.authSessionRenewMs, + ...(overrides.now ? { now: overrides.now } : {}), + }); + const adminService = new AdminService({ + users: usersRepo, + authSessions: authSessionsRepo, + projects: projectsRepo, + projectService, + ...(overrides.now ? { now: overrides.now } : {}), + }); + const sessionService = new SessionService({ + root: config.root, + sessions: sessionsRepo, + manager, + projectConfig: projectConfigService, + }); + // Schedule scheduler: active only while the server is running. Only + // assembled here; start() is called in index.ts (tests drive it via tickOnce, no real timer). + const scheduler = new Scheduler({ + root: config.root, + repo: schedulesRepo, + projects: projectsRepo, + sessions: sessionsRepo, + runner: manager, + sessionCreator: sessionService, + errors, + notify: (userId, event) => { + channels.get(userChannelKey(userId)).publish(event, "server_event"); + }, + ...(overrides.now ? { now: () => overrides.now!().getTime() } : {}), + }); + + return { + config, + db, + sessionsRepo, + prefsRepo, + authService, + adminService, + projectService, + projectConfigService, + agentService, + agentConfigService, + sessionService, + traceService, + usageService, + workspaceFiles, + benchmarks, + snapshots, + schedulesRepo, + scheduler, + channels, + manager, + errors, + log, + }; +} + +/** Assembles the Hono app (does not listen on a port). */ +export function createApp(deps: AppDeps): Hono { + const app = new Hono(); + + // Error recording is layered in a lambda wrapping onError: handleError stays a + // pure function with unchanged behavior (HttpError is mapped as-is, unknown + // exceptions are logged with a stack trace and collapsed to 500), and recording + // to the DB is just a side-effect layered on top. + app.onError((err, c) => { + const projectId = attributedProjectId(c, deps); + deps.errors.record({ + source: "http", + err, + ...(projectId !== undefined ? { ctx: { projectId } } : {}), + }); + return handleError(err, c); + }); + app.notFound((c) => c.json(errorBody("not_found", "接口不存在。"), 404)); + + // Request logging: a minimal one-liner (method path status ms). + app.use("*", async (c, next) => { + const start = performance.now(); + await next(); + const ms = Math.round(performance.now() - start); + deps.log(`${c.req.method} ${c.req.path} ${c.res.status} ${ms}ms`); + }); + + // API common defenses: request body size cap (20MB) and write-request Content-Type (one of the CSRF MVP defenses). + app.use("/api/*", async (c, next) => { + const contentLength = Number(c.req.header("content-length") ?? 0); + if (contentLength > MAX_BODY_BYTES) { + throw new HttpError(413, "payload_too_large", "请求体超过 20MB 上限。"); + } + await next(); + }); + app.use("/api/*", jsonOnlyWrites); + + // Public routes (no login required). + app.route("/api/auth", authRoutes(deps)); + + // Protected routes: cookie -> auth_session -> user. + const auth = authMiddleware(deps.authService); + app.use("/api/*", auth); + app.route("/api/me", meRoutes(deps)); + app.route("/api/admin/users", adminUsersRoutes(deps)); + app.route("/api/events", eventsRoutes(deps)); + // Skill library listing: readable once logged in, not nested under a Project prefix. + app.route("/api/skills", skillLibraryRoutes()); + app.route("/api/projects", projectsRoutes(deps)); + app.route("/api/projects/:projectId/members", membersRoutes(deps)); + app.route("/api/projects/:projectId/models", modelsRoutes(deps)); + app.route("/api/projects/:projectId/agents", agentsRoutes(deps)); + app.route("/api/projects/:projectId/dirs", dirsRoutes(deps)); + app.route("/api/projects/:projectId/agents/:agentId/config", agentConfigRoutes(deps)); + app.route("/api/projects/:projectId/agents/:agentId/vault", vaultRoutes(deps)); + app.route("/api/projects/:projectId/agents/:agentId/schedules", scheduleRoutes(deps)); + app.route("/api/projects/:projectId/agents/:agentId/benchmarks", benchmarksRoutes(deps)); + app.route("/api/projects/:projectId/agents/:agentId/skills", agentSkillsRoutes(deps)); + app.route("/api/projects/:projectId/agents/:agentId", agentTransferRoutes(deps)); + app.route("/api/projects/:projectId/agents/:agentId/traces", agentTracesRoutes(deps)); + app.route("/api/projects/:projectId/agents/:agentId/sessions", agentSessionsRoutes(deps)); + app.route("/api/projects/:projectId/usage", usageRoutes(deps)); + app.route("/api/sessions", sessionsRoutes(deps)); + + // Static hosting (production): serves the frontend build output when webDist exists, with SPA fallback to index.html. + if (fs.existsSync(deps.config.webDist)) { + registerStaticRoutes(app, deps.config.webDist); + } + + return app; +} + +/** + * The Project an error is attributed to: only + * attributed when the URL has a `:projectId` **and** the requester genuinely has + * access to that Project; otherwise recorded as unattributed (`project_id IS + * NULL`, visible only to admins). + * + * onError also has to handle requests that **haven't passed permission checks + * yet** — a 401 from being logged out, a 404 from not being a member, both get + * recorded here. Attributing directly from the URL parameter would let anyone + * (not necessarily a member of that Project, or even logged in) pick a projectId + * and hammer it repeatedly to pollute another user's Project with error stats. + * Traces that can't be attributed simply fall into the admin view (unattributed + * errors are only visible to admins by design anyway), which is exactly where + * unauthorized probing belongs. + * + * Two defenses here, because this code runs on the error-handling path: + * - `c.var.user`'s static type is non-null, but authMiddleware never sets it + * **before** throwing the 401 when logged out, so at runtime it may actually be + * undefined — it can only be read safely, never destructured directly. + * - Exceptions are swallowed entirely: throwing here would break onError itself + * (possibly recursively); any judgment failure falls back to unattributed. + */ +function attributedProjectId(c: Context, deps: AppDeps): string | undefined { + try { + const projectId = c.req.param("projectId"); + if (projectId === undefined) return undefined; + const user = c.get("user") as UserRow | undefined; + if (user === undefined) return undefined; + return deps.projectService.canAccess(user.userId, projectId) ? projectId : undefined; + } catch { + return undefined; + } +} + +const CONTENT_TYPES: Record = { + ".html": "text/html; charset=utf-8", + ".js": "text/javascript; charset=utf-8", + ".css": "text/css; charset=utf-8", + ".json": "application/json", + ".svg": "image/svg+xml", + ".png": "image/png", + ".jpg": "image/jpeg", + ".ico": "image/x-icon", + ".map": "application/json", + ".txt": "text/plain; charset=utf-8", + ".woff2": "font/woff2", +}; + +/** Minimal static file server (avoiding an extra dependency): path traversal protection + SPA fallback. */ +function registerStaticRoutes(app: Hono, webDist: string): void { + app.get("*", async (c) => { + const reqPath = decodeURIComponent(c.req.path); + if (reqPath.startsWith("/api/")) { + return c.json(errorBody("not_found", "接口不存在。"), 404); + } + const rel = reqPath.replace(/^\/+/, ""); + const resolved = path.resolve(webDist, rel === "" ? "index.html" : rel); + // Guard against path traversal: once resolved, it must still be inside webDist. + const base = path.resolve(webDist); + const target = + resolved === base || resolved.startsWith(base + path.sep) + ? resolved + : path.join(base, "index.html"); + let file = target; + try { + const stat = await fsp.stat(file); + if (stat.isDirectory()) file = path.join(file, "index.html"); + await fsp.access(file); + } catch { + file = path.join(base, "index.html"); // SPA fallback + } + let content: Buffer; + try { + content = await fsp.readFile(file); + } catch { + return c.json(errorBody("not_found", "资源不存在。"), 404); + } + const type = CONTENT_TYPES[path.extname(file).toLowerCase()] ?? "application/octet-stream"; + return new Response(new Uint8Array(content), { + status: 200, + headers: { "Content-Type": type }, + }); + }); +} diff --git a/packages/server/src/auth/middleware.ts b/packages/server/src/auth/middleware.ts new file mode 100644 index 0000000..1258889 --- /dev/null +++ b/packages/server/src/auth/middleware.ts @@ -0,0 +1,58 @@ +/** + * Auth middleware: cookie -> auth_session -> + * user injected into c.var. + * + * Accessing a protected API while logged out -> 401 `{error:{code:"unauthorized"}}`. + * CSRF (MVP): SameSite=Lax cookie + write requests only accept + * `Content-Type: application/json` (an HTML form can't forge that Content-Type), + * see the README security notes. + */ +import type { MiddlewareHandler } from "hono"; +import { getCookie } from "hono/cookie"; +import { HttpError } from "../http/errors.js"; +import type { UserRow } from "../db/repos/users.js"; +import type { AuthService } from "./service.js"; + +/** Session cookie name. */ +export const SESSION_COOKIE = "penguin_session"; + +/** Hono env: variables injected by the auth middleware. */ +export type AppEnv = { + Variables: { + user: UserRow; + }; +}; + +/** Gets the current user (available after authMiddleware). */ +export function currentUser(c: { var: { user: UserRow } }): UserRow { + return c.var.user; +} + +export function authMiddleware(auth: AuthService): MiddlewareHandler { + return async (c, next) => { + const token = getCookie(c, SESSION_COOKIE); + const user = token ? auth.authenticate(token) : null; + if (!user) { + throw new HttpError(401, "unauthorized", "未登录或登录已过期。"); + } + c.set("user", user); + await next(); + }; +} + +const WRITE_METHODS = new Set(["POST", "PUT", "PATCH", "DELETE"]); + +/** + * Content-Type defense for write requests: a write request with a Content-Type + * other than application/json is rejected (a request with no Content-Type and an + * empty body is let through — an HTML form always carries a form-type Content-Type). + */ +export const jsonOnlyWrites: MiddlewareHandler = async (c, next) => { + if (WRITE_METHODS.has(c.req.method)) { + const contentType = c.req.header("content-type"); + if (contentType && !contentType.toLowerCase().startsWith("application/json")) { + throw new HttpError(415, "unsupported_media_type", "写请求仅接受 application/json。"); + } + } + await next(); +}; diff --git a/packages/server/src/auth/password.ts b/packages/server/src/auth/password.ts new file mode 100644 index 0000000..3918fb6 --- /dev/null +++ b/packages/server/src/auth/password.ts @@ -0,0 +1,72 @@ +/** + * Password hashing (slow-hash storage). + * + * Uses node:crypto's scrypt (built-in, no extra dependency, meets the same + * slow-hash requirement as bcrypt/argon2). Storage format: + * `scrypt$N$r$p$$` — parameters are stored alongside the hash, + * so old hashes remain verifiable after future parameter tuning; comparison uses + * timingSafeEqual to guard against timing side-channels. + */ +import { randomBytes, scrypt, timingSafeEqual } from "node:crypto"; + +const SCRYPT_N = 16384; +const SCRYPT_R = 8; +const SCRYPT_P = 1; +const SALT_BYTES = 16; +const KEY_BYTES = 64; + +function scryptAsync( + password: string, + salt: Buffer, + keyLen: number, + n: number, + r: number, + p: number, +): Promise { + return new Promise((resolve, reject) => { + scrypt(password, salt, keyLen, { N: n, r, p, maxmem: 128 * 1024 * 1024 }, (err, key) => { + if (err) reject(err); + else resolve(key); + }); + }); +} + +/** Generates a password hash in `scrypt$N$r$p$salt$hash` format. */ +export async function hashPassword(password: string): Promise { + const salt = randomBytes(SALT_BYTES); + const key = await scryptAsync(password, salt, KEY_BYTES, SCRYPT_N, SCRYPT_R, SCRYPT_P); + return [ + "scrypt", + String(SCRYPT_N), + String(SCRYPT_R), + String(SCRYPT_P), + salt.toString("base64"), + key.toString("base64"), + ].join("$"); +} + +/** Verifies a password; returns false if the stored string has an invalid format (never throws, so the login path can uniformly treat it as a credential error). */ +export async function verifyPassword(password: string, stored: string): Promise { + const parts = stored.split("$"); + if (parts.length !== 6 || parts[0] !== "scrypt") return false; + const n = Number(parts[1]); + const r = Number(parts[2]); + const p = Number(parts[3]); + if (!Number.isInteger(n) || !Number.isInteger(r) || !Number.isInteger(p)) return false; + let salt: Buffer; + let expected: Buffer; + try { + salt = Buffer.from(parts[4]!, "base64"); + expected = Buffer.from(parts[5]!, "base64"); + } catch { + return false; + } + if (salt.length === 0 || expected.length === 0) return false; + let actual: Buffer; + try { + actual = await scryptAsync(password, salt, expected.length, n, r, p); + } catch { + return false; + } + return actual.length === expected.length && timingSafeEqual(actual, expected); +} diff --git a/packages/server/src/auth/service.ts b/packages/server/src/auth/service.ts new file mode 100644 index 0000000..d0291d3 --- /dev/null +++ b/packages/server/src/auth/service.ts @@ -0,0 +1,137 @@ +/** + * Auth service: built-in admin seeding / + * login / logout / password change / session validation. + * + * - No open registration: on startup, if there are no users at all, the built-in + * admin `admin` is seeded (initial password admin123), and it adopts + * `default_project`; all other users are created by an admin via the user + * backend (admin-service). + * - An initial password (whether seeded or set by an admin) is flagged with + * password_is_initial, which the frontend uses to prompt for a password change soon. + * - Sessions: a 32-byte random token, with only its sha256 hash stored in the DB; + * valid for 7 days, with sliding renewal once less than 6 days remain. + */ +import { createHash, randomBytes } from "node:crypto"; +import type { UserInfo } from "../api/types.js"; +import { HttpError } from "../http/errors.js"; +import type { AuthSessionsRepo } from "../db/repos/auth-sessions.js"; +import type { UserRow, UsersRepo } from "../db/repos/users.js"; +import { hashPassword, verifyPassword } from "./password.js"; + +export const MIN_PASSWORD_LENGTH = 8; + +/** Built-in admin: user_id and initial password (matches the README and login-page hint). */ +export const ADMIN_USER_ID = "admin"; +export const ADMIN_INITIAL_PASSWORD = "admin123"; + +function sha256Hex(value: string): string { + return createHash("sha256").update(value).digest("hex"); +} + +export function toUserInfo(row: UserRow): UserInfo { + return { + userId: row.userId, + isAdmin: row.isAdmin, + passwordIsInitial: row.passwordIsInitial, + createdAt: row.createdAt, + }; +} + +export interface AuthServiceDeps { + users: UsersRepo; + authSessions: AuthSessionsRepo; + /** Provisions the initial Project at signup (injected by project-service, to avoid a circular dependency). */ + provisionInitialProject: (user: UserRow, isAdmin: boolean) => Promise; + sessionTtlMs: number; + sessionRenewMs: number; + now?: () => Date; +} + +export class AuthService { + private readonly now: () => Date; + + constructor(private readonly deps: AuthServiceDeps) { + this.now = deps.now ?? (() => new Date()); + } + + /** + * Startup seeding (idempotent): creates the built-in admin and adopts + * default_project when the users table is empty; if the initial Project fails, + * the user row is rolled back and the server retries on next startup. + */ + async seedAdmin(): Promise { + if (this.deps.users.count() > 0) return; + const user: UserRow = { + userId: ADMIN_USER_ID, + passwordHash: await hashPassword(ADMIN_INITIAL_PASSWORD), + isAdmin: true, + passwordIsInitial: true, + createdAt: this.now().toISOString(), + }; + this.deps.users.insert(user); + try { + await this.deps.provisionInitialProject(user, true); + } catch (err) { + this.deps.users.delete(user.userId); + throw err; + } + } + + async login(userId: string, password: string): Promise<{ user: UserInfo; token: string }> { + const row = this.deps.users.findById(userId); + const ok = row !== null && (await verifyPassword(password, row.passwordHash)); + if (!row || !ok) { + throw new HttpError(401, "invalid_credentials", "用户名或密码错误。"); + } + this.deps.authSessions.deleteExpired(this.now().toISOString()); + return { user: toUserInfo(row), token: this.issueSession(row.userId) }; + } + + /** Self password change (user settings): validates the old password, and on success clears the initial-password flag; the current session remains valid. */ + async changePassword(userId: string, oldPassword: string, newPassword: string): Promise { + const row = this.deps.users.findById(userId); + if (!row || !(await verifyPassword(oldPassword, row.passwordHash))) { + throw new HttpError(400, "password_mismatch", "当前密码不正确。"); + } + if (newPassword.length < MIN_PASSWORD_LENGTH) { + throw new HttpError(400, "invalid_password", "密码至少 8 个字符。"); + } + this.deps.users.updatePassword(userId, await hashPassword(newPassword), false); + } + + logout(token: string): void { + this.deps.authSessions.delete(sha256Hex(token)); + } + + /** Validates the cookie token: returns null if expired/unknown; sliding renewal once less than 6 days remain. */ + authenticate(token: string): UserRow | null { + const tokenHash = sha256Hex(token); + const session = this.deps.authSessions.findByTokenHash(tokenHash); + if (!session) return null; + const now = this.now(); + const expiresAt = Date.parse(session.expiresAt); + if (!(expiresAt > now.getTime())) { + this.deps.authSessions.delete(tokenHash); + return null; + } + if (expiresAt - now.getTime() < this.deps.sessionRenewMs) { + this.deps.authSessions.touch( + tokenHash, + new Date(now.getTime() + this.deps.sessionTtlMs).toISOString(), + ); + } + return this.deps.users.findById(session.userId); + } + + private issueSession(userId: string): string { + const token = randomBytes(32).toString("base64url"); + const now = this.now(); + this.deps.authSessions.insert({ + tokenHash: sha256Hex(token), + userId, + createdAt: now.toISOString(), + expiresAt: new Date(now.getTime() + this.deps.sessionTtlMs).toISOString(), + }); + return token; + } +} diff --git a/packages/server/src/config.ts b/packages/server/src/config.ts new file mode 100644 index 0000000..d5f9cb1 --- /dev/null +++ b/packages/server/src/config.ts @@ -0,0 +1,68 @@ +/** + * Server runtime config (ServerConfig) — parsed from environment variables. + * + * The data root directory is shared with the SDK / CLI (`resolveRoot()`: + * PENGUIN_HOME or ~/.penguin/data); the SQLite index database defaults to + * `/web.db` (overridable via PENGUIN_WEB_DB, tests use ":memory:"). + * In production, the SPA is served statically once the frontend build output + * directory (PENGUIN_WEB_DIST, the bundled web-dist/, or ../web/dist) is + * detected to exist. + * Docs: /docs/configuration § "Environment variables". + */ +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { resolveRoot } from "@prismshadow/penguin-core"; + +export interface ServerConfig { + /** Local data root directory (shared with the SDK/CLI). */ + root: string; + /** HTTP listen address and port (defaults to 127.0.0.1:7364, deliberately avoiding common ports like 3000/8080). */ + host: string; + port: number; + /** SQLite database path; ":memory:" for test injection. */ + dbPath: string; + /** Frontend static assets directory; whether it's enabled is decided by checking existence when the app is assembled. */ + webDist: string; + /** Login session validity period (7 days). */ + authSessionTtlMs: number; + /** Sliding renewal threshold: if the remaining validity is below this value when validation succeeds, it's renewed to the full TTL (renews under 6 days). */ + authSessionRenewMs: number; +} + +const DAY_MS = 24 * 60 * 60 * 1000; + +/** + * Default frontend build output directory, first match wins: + * - `/web-dist`: npm package layout — the release workflow copies the built + * web assets into the published package, so an `npm install` gets the Web UI too; + * - `/../web/dist`: monorepo layout (resolves the same whether running + * from src or dist), also the fallback when neither exists. + */ +function defaultWebDist(): string { + const here = path.dirname(fileURLToPath(import.meta.url)); + const bundled = path.resolve(here, "..", "web-dist"); + if (fs.existsSync(bundled)) return bundled; + return path.resolve(here, "..", "..", "web", "dist"); +} + +/** Parses server config from environment variables (PORT / HOST / PENGUIN_HOME / PENGUIN_WEB_DIST / PENGUIN_WEB_DB). */ +export function resolveServerConfig(env: NodeJS.ProcessEnv = process.env): ServerConfig { + const root = env.PENGUIN_HOME ?? resolveRoot(); + // An empty PORT string is treated as unset (the common `.env` case of an empty + // `PORT=`): Number("") === 0 would pass the range check and bind to a random + // port; this matches the CLI's resolvePort convention. + const port = Number(env.PORT || 7364); + if (!Number.isInteger(port) || port < 0 || port > 65535) { + throw new Error(`非法端口配置 PORT=${env.PORT}`); + } + return { + root, + host: env.HOST ?? "127.0.0.1", + port, + dbPath: env.PENGUIN_WEB_DB ?? path.join(root, "web.db"), + webDist: env.PENGUIN_WEB_DIST ?? defaultWebDist(), + authSessionTtlMs: 7 * DAY_MS, + authSessionRenewMs: 6 * DAY_MS, + }; +} diff --git a/packages/server/src/db/database.ts b/packages/server/src/db/database.ts new file mode 100644 index 0000000..6fd4467 --- /dev/null +++ b/packages/server/src/db/database.ts @@ -0,0 +1,28 @@ +/** + * SQLite connection & initialization (node:sqlite DatabaseSync). + * + * Single process, single writer: a synchronous API is sufficient and avoids a connection + * pool; WAL mode and foreign key constraints are enabled. Table-creation SQL runs on open + * (idempotent), with no migration branches (product not yet released). + */ +import { mkdirSync } from "node:fs"; +import path from "node:path"; +import type { DatabaseSync } from "node:sqlite"; +import { SCHEMA_SQL } from "./schema.js"; + +// Fetch the runtime module via process.getBuiltinModule (node >=22.3): avoids static +// resolution of `node:sqlite` by bundlers/vite (some tools' builtin lists don't yet +// recognize this experimental module). +const sqlite = process.getBuiltinModule("node:sqlite"); + +/** Open (creating if necessary) the database: ensure the parent directory exists, set PRAGMAs, run table creation. */ +export function openDatabase(dbPath: string): DatabaseSync { + if (dbPath !== ":memory:") { + mkdirSync(path.dirname(dbPath), { recursive: true }); + } + const db = new sqlite.DatabaseSync(dbPath); + db.exec("PRAGMA journal_mode = WAL;"); + db.exec("PRAGMA foreign_keys = ON;"); + db.exec(SCHEMA_SQL); + return db; +} diff --git a/packages/server/src/db/repos/agents.ts b/packages/server/src/db/repos/agents.ts new file mode 100644 index 0000000..197f1b4 --- /dev/null +++ b/packages/server/src/db/repos/agents.ts @@ -0,0 +1,49 @@ +/** + * agents table repo: Agent index; name/description live in system_config.yaml. + */ +import type { DatabaseSync } from "node:sqlite"; + +export interface AgentRow { + projectId: string; + agentId: string; + createdAt: string; +} + +export class AgentsRepo { + constructor(private readonly db: DatabaseSync) {} + + /** Idempotent insert: backfills an untracked Agent discovered via directory scan; shared with explicit creation. */ + insertOrIgnore(row: AgentRow): void { + this.db + .prepare("INSERT OR IGNORE INTO agents (project_id, agent_id, created_at) VALUES (?, ?, ?)") + .run(row.projectId, row.agentId, row.createdAt); + } + + exists(projectId: string, agentId: string): boolean { + const r = this.db + .prepare("SELECT 1 AS x FROM agents WHERE project_id = ? AND agent_id = ?") + .get(projectId, agentId); + return r !== undefined; + } + + list(projectId: string): AgentRow[] { + const rows = this.db + .prepare("SELECT * FROM agents WHERE project_id = ? ORDER BY created_at ASC, agent_id ASC") + .all(projectId); + return rows.map((r) => ({ + projectId: r.project_id as string, + agentId: r.agent_id as string, + createdAt: r.created_at as string, + })); + } + + delete(projectId: string, agentId: string): void { + this.db + .prepare("DELETE FROM agents WHERE project_id = ? AND agent_id = ?") + .run(projectId, agentId); + } + + deleteByProject(projectId: string): void { + this.db.prepare("DELETE FROM agents WHERE project_id = ?").run(projectId); + } +} diff --git a/packages/server/src/db/repos/auth-sessions.ts b/packages/server/src/db/repos/auth-sessions.ts new file mode 100644 index 0000000..a7b4a4b --- /dev/null +++ b/packages/server/src/db/repos/auth-sessions.ts @@ -0,0 +1,57 @@ +/** + * auth_sessions table repo (server-side sessions backing the HttpOnly cookie). + * + * Stores only the sha256(token) hex hash; the raw token appears only in the cookie. + */ +import type { DatabaseSync } from "node:sqlite"; + +export interface AuthSessionRow { + tokenHash: string; + userId: string; + createdAt: string; + expiresAt: string; +} + +export class AuthSessionsRepo { + constructor(private readonly db: DatabaseSync) {} + + insert(row: AuthSessionRow): void { + this.db + .prepare( + "INSERT INTO auth_sessions (token_hash, user_id, created_at, expires_at) VALUES (?, ?, ?, ?)", + ) + .run(row.tokenHash, row.userId, row.createdAt, row.expiresAt); + } + + findByTokenHash(tokenHash: string): AuthSessionRow | null { + const r = this.db.prepare("SELECT * FROM auth_sessions WHERE token_hash = ?").get(tokenHash); + if (!r) return null; + return { + tokenHash: r.token_hash as string, + userId: r.user_id as string, + createdAt: r.created_at as string, + expiresAt: r.expires_at as string, + }; + } + + /** Sliding renewal: update the expiration time. */ + touch(tokenHash: string, expiresAt: string): void { + this.db + .prepare("UPDATE auth_sessions SET expires_at = ? WHERE token_hash = ?") + .run(expiresAt, tokenHash); + } + + delete(tokenHash: string): void { + this.db.prepare("DELETE FROM auth_sessions WHERE token_hash = ?").run(tokenHash); + } + + /** Opportunistically clean up expired sessions (called during login/validation). */ + deleteExpired(nowIso: string): void { + this.db.prepare("DELETE FROM auth_sessions WHERE expires_at < ?").run(nowIso); + } + + /** Clear all sessions for a user (forces re-login after an admin resets the password). */ + deleteByUser(userId: string): void { + this.db.prepare("DELETE FROM auth_sessions WHERE user_id = ?").run(userId); + } +} diff --git a/packages/server/src/db/repos/errors.ts b/packages/server/src/db/repos/errors.ts new file mode 100644 index 0000000..57d9993 --- /dev/null +++ b/packages/server/src/db/repos/errors.ts @@ -0,0 +1,219 @@ +/** + * error_records table repo: one row per error the server catches. + * + * Key difference from usage_records: **all attribution columns are nullable** — errors + * from the login/register endpoints have no Project, and process-level catch-alls + * (uncaughtException) don't even have a request. These **unattributed errors are visible + * only to admins** (`ErrorFilter.includeGlobal`, defaults to false): they represent other + * users' login failures, misdirected Session access, and process crashes (the message may + * contain internal paths and variable values), so surfacing them in any regular member's + * statistics center would be a cross-tenant information leak — regular members see only + * their own Project's errors. Admins still see the category that most needs visibility. + * + * **Row cap** (MAX_ROWS): errors often come in storms (an API scan producing a wall of + * 404s, a tool failing repeatedly in a loop), and an uncapped table would blow up disk + * usage. So the insert path enforces the cap: check capacity every PRUNE_EVERY inserts + * (insertion sits on the error-handling path, so we avoid COUNT on every call), and when + * over the limit, evict the oldest rows in ascending id order. The excess is computed via + * an **exact COUNT**, not an approximation like `id <= MAX(id) - :max` — after + * deleteByProject removes rows, id and row count diverge, and the approximation would + * wrongly delete still-valid data within the cap. The first line of defense is + * ErrorRecorder's short-window deduplication. + */ +import type { DatabaseSync } from "node:sqlite"; + +/** Row cap: once exceeded, oldest rows are evicted in ascending id order (see file header). */ +export const MAX_ROWS = 20000; + +/** Check capacity every N inserts (see file header: insertion sits on the error-handling path, so we avoid COUNT on every call). */ +export const PRUNE_EVERY = 200; + +/** Capacity parameters (default to the two constants above; tests inject small values to exercise the eviction path). */ +export interface ErrorsRepoLimits { + maxRows?: number; + pruneEvery?: number; +} + +export interface ErrorRecordInsert { + ts: string; + date: string; + projectId: string | null; + agentId: string | null; + sessionId: string | null; + source: string; + /** expected (HttpError, business 4xx) | unexpected (500 / unforeseen runtime error). */ + kind: string; + code: string; + status: number | null; + message: string; +} + +/** Generic filter: date range + agent (errors have no Model dimension, so no model filter). */ +export interface ErrorFilter { + from?: string; + to?: string; + agentId?: string; + /** Whether to include unattributed errors (`project_id IS NULL`): admins only, defaults to false (see file header). */ + includeGlobal?: boolean; +} + +/** Total error count and how many are unexpected (stats for the statistics center's error panel). */ +export interface ErrorSummary { + total: number; + unexpected: number; +} + +/** Occurrence count for one source · code pair (the error panel's "most common" metric). */ +export interface ErrorCodeCount { + source: string; + code: string; + kind: string; + count: number; +} + +/** One error summary row (a row in the error panel's table). */ +export interface ErrorItem { + ts: string; + source: string; + code: string; + kind: string; + message: string; +} + +export class ErrorsRepo { + private readonly maxRows: number; + private readonly pruneEvery: number; + /** Insert count since the last capacity check (see file header: avoids COUNT on every call). */ + private sinceCheck = 0; + + constructor( + private readonly db: DatabaseSync, + limits: ErrorsRepoLimits = {}, + ) { + this.maxRows = limits.maxRows ?? MAX_ROWS; + this.pruneEvery = limits.pruneEvery ?? PRUNE_EVERY; + } + + insert(r: ErrorRecordInsert): void { + this.db + .prepare( + `INSERT INTO error_records + (ts, date, project_id, agent_id, session_id, source, kind, code, status, message) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + ) + .run( + r.ts, + r.date, + r.projectId, + r.agentId, + r.sessionId, + r.source, + r.kind, + r.code, + r.status, + r.message, + ); + if (++this.sinceCheck >= this.pruneEvery) { + this.sinceCheck = 0; + this.pruneOverflow(); + } + } + + /** Capacity enforcement (see file header): compute the exact excess via COUNT, then delete the oldest rows in ascending id order. */ + private pruneOverflow(): void { + const row = this.db.prepare("SELECT COUNT(*) AS n FROM error_records").get()!; + const excess = (row.n as number) - this.maxRows; + if (excess <= 0) return; + this.db + .prepare( + `DELETE FROM error_records WHERE id IN ( + SELECT id FROM error_records ORDER BY id ASC LIMIT :excess + )`, + ) + .run({ excess }); + } + + /** WHERE fragment and named params: this Project (admins additionally get unattributed errors), plus optional date/agent filter. */ + private conds( + projectId: string, + f: ErrorFilter, + ): { where: string; params: Record } { + // Unattributed errors (login failures, process crashes, ...) are visible only to admins; otherwise it's a cross-tenant leak — see file header. + const conds = [ + f.includeGlobal === true ? "(project_id = :pid OR project_id IS NULL)" : "project_id = :pid", + ]; + const params: Record = { pid: projectId }; + if (f.from !== undefined) { + conds.push("date >= :from"); + params.from = f.from; + } + if (f.to !== undefined) { + conds.push("date <= :to"); + params.to = f.to; + } + if (f.agentId !== undefined) { + // Filtering by Agent naturally leaves only that Agent's errors (HTTP / process-level errors have no agent_id). + conds.push("agent_id = :agentId"); + params.agentId = f.agentId; + } + return { where: conds.join(" AND "), params }; + } + + /** Total count + how many are unexpected. */ + summary(projectId: string, f: ErrorFilter = {}): ErrorSummary { + const { where, params } = this.conds(projectId, f); + const row = this.db + .prepare( + `SELECT COUNT(*) AS total, + COALESCE(SUM(CASE WHEN kind = 'unexpected' THEN 1 ELSE 0 END), 0) AS unexpected + FROM error_records WHERE ${where}`, + ) + .get(params)!; + return { total: row.total as number, unexpected: row.unexpected as number }; + } + + /** The most frequent source · code (ties broken by code's lexicographic order); null if there are no errors. */ + topCode(projectId: string, f: ErrorFilter = {}): ErrorCodeCount | null { + const { where, params } = this.conds(projectId, f); + const row = this.db + .prepare( + `SELECT source, code, kind, COUNT(*) AS count + FROM error_records WHERE ${where} + GROUP BY source, code, kind + ORDER BY count DESC, code ASC + LIMIT 1`, + ) + .get(params); + if (!row) return null; + return { + source: row.source as string, + code: row.code as string, + kind: row.kind as string, + count: row.count as number, + }; + } + + /** The most recent `limit` entries (reverse chronological order). */ + recent(projectId: string, f: ErrorFilter = {}, limit = 20): ErrorItem[] { + const { where, params } = this.conds(projectId, f); + const rows = this.db + .prepare( + `SELECT ts, source, code, kind, message + FROM error_records WHERE ${where} + ORDER BY id DESC LIMIT :limit`, + ) + .all({ ...params, limit }); + return rows.map((r) => ({ + ts: r.ts as string, + source: r.source as string, + code: r.code as string, + kind: r.kind as string, + message: r.message as string, + })); + } + + /** Cascading cleanup on Project deletion (unattributed errors belong to no Project and are unaffected). */ + deleteByProject(projectId: string): void { + this.db.prepare("DELETE FROM error_records WHERE project_id = ?").run(projectId); + } +} diff --git a/packages/server/src/db/repos/members.ts b/packages/server/src/db/repos/members.ts new file mode 100644 index 0000000..36f6410 --- /dev/null +++ b/packages/server/src/db/repos/members.ts @@ -0,0 +1,47 @@ +/** + * Repo for the project_members table: only member + * authorization relationships — the owner is never in this table. + */ +import type { DatabaseSync } from "node:sqlite"; + +export interface MemberRow { + projectId: string; + userId: string; + createdAt: string; +} + +export class MembersRepo { + constructor(private readonly db: DatabaseSync) {} + + insert(row: MemberRow): void { + this.db + .prepare("INSERT INTO project_members (project_id, user_id, created_at) VALUES (?, ?, ?)") + .run(row.projectId, row.userId, row.createdAt); + } + + isMember(projectId: string, userId: string): boolean { + const r = this.db + .prepare("SELECT 1 AS x FROM project_members WHERE project_id = ? AND user_id = ?") + .get(projectId, userId); + return r !== undefined; + } + + list(projectId: string): MemberRow[] { + const rows = this.db + .prepare( + "SELECT project_id, user_id, created_at FROM project_members WHERE project_id = ? ORDER BY created_at ASC", + ) + .all(projectId); + return rows.map((r) => ({ + projectId: r.project_id as string, + userId: r.user_id as string, + createdAt: r.created_at as string, + })); + } + + delete(projectId: string, userId: string): void { + this.db + .prepare("DELETE FROM project_members WHERE project_id = ? AND user_id = ?") + .run(projectId, userId); + } +} diff --git a/packages/server/src/db/repos/projects.ts b/packages/server/src/db/repos/projects.ts new file mode 100644 index 0000000..5053d74 --- /dev/null +++ b/packages/server/src/db/repos/projects.ts @@ -0,0 +1,78 @@ +/** + * Repo for the projects table: an index of ownership + * relationships; the display name lives in project_config.toml. + */ +import type { DatabaseSync } from "node:sqlite"; +import type { ProjectRole } from "../../api/types.js"; + +export interface ProjectRow { + projectId: string; + ownerUserId: string; + createdAt: string; +} + +export interface AccessibleProjectRow extends ProjectRow { + role: ProjectRole; +} + +function mapRow(r: Record): ProjectRow { + return { + projectId: r.project_id as string, + ownerUserId: r.owner_user_id as string, + createdAt: r.created_at as string, + }; +} + +export class ProjectsRepo { + constructor(private readonly db: DatabaseSync) {} + + insert(row: ProjectRow): void { + this.db + .prepare("INSERT INTO projects (project_id, owner_user_id, created_at) VALUES (?, ?, ?)") + .run(row.projectId, row.ownerUserId, row.createdAt); + } + + findById(projectId: string): ProjectRow | null { + const r = this.db.prepare("SELECT * FROM projects WHERE project_id = ?").get(projectId); + return r ? mapRow(r) : null; + } + + /** All Projects (used by the scheduler's reconciliation scan), ascending by creation time. */ + listAll(): ProjectRow[] { + const rows = this.db + .prepare("SELECT * FROM projects ORDER BY created_at ASC, project_id ASC") + .all(); + return rows.map((r) => mapRow(r as Record)); + } + + /** Projects owned by or shared with the current user (including their role), ascending by creation time. */ + listAccessible(userId: string): AccessibleProjectRow[] { + const rows = this.db + .prepare( + `SELECT p.project_id, p.owner_user_id, p.created_at, + CASE WHEN p.owner_user_id = :uid THEN 'owner' ELSE 'member' END AS role + FROM projects p + WHERE p.owner_user_id = :uid + OR EXISTS (SELECT 1 FROM project_members m + WHERE m.project_id = p.project_id AND m.user_id = :uid) + ORDER BY p.created_at ASC, p.project_id ASC`, + ) + .all({ uid: userId }); + return rows.map((r) => ({ + ...mapRow(r), + role: r.role as ProjectRole, + })); + } + + /** All Projects owned by a given user (used for cascading cleanup when an admin deletes a user). */ + listByOwner(userId: string): ProjectRow[] { + const rows = this.db + .prepare("SELECT * FROM projects WHERE owner_user_id = ? ORDER BY created_at ASC") + .all(userId); + return rows.map(mapRow); + } + + delete(projectId: string): void { + this.db.prepare("DELETE FROM projects WHERE project_id = ?").run(projectId); + } +} diff --git a/packages/server/src/db/repos/schedules.ts b/packages/server/src/db/repos/schedules.ts new file mode 100644 index 0000000..d743ce7 --- /dev/null +++ b/packages/server/src/db/repos/schedules.ts @@ -0,0 +1,191 @@ +/** + * Repo for schedule runtime state: intent and state are separate — the file is + * declarative intent, this table only records runtime state such as + * "fired before / last fired / missed / disabled" plus the creator. + * + * Identity rule: a change to `start_at` is treated as a new task instance + * (registerOrSync resets the trigger state); a change to the file content fingerprint + * only clears the disabled flag (the file becomes effective again after reconciliation). + */ +import type { DatabaseSync } from "node:sqlite"; + +export interface ScheduleStateRow { + projectId: string; + agentId: string; + name: string; + creatorUserId: string | null; + startAtMs: number; + defHash: string; + lastSlotMs: number | null; + lastFiredAt: string | null; + firedOnce: boolean; + missed: boolean; + invalidReason: string | null; +} + +function mapRow(r: Record): ScheduleStateRow { + return { + projectId: r.project_id as string, + agentId: r.agent_id as string, + name: r.name as string, + creatorUserId: (r.creator_user_id as string | null) ?? null, + startAtMs: Number(r.start_at_ms), + defHash: r.def_hash as string, + lastSlotMs: r.last_slot_ms === null ? null : Number(r.last_slot_ms), + lastFiredAt: (r.last_fired_at as string | null) ?? null, + firedOnce: Number(r.fired_once) === 1, + missed: Number(r.missed) === 1, + invalidReason: (r.invalid_reason as string | null) ?? null, + }; +} + +export class SchedulesRepo { + constructor(private readonly db: DatabaseSync) {} + + find(projectId: string, agentId: string, name: string): ScheduleStateRow | null { + const r = this.db + .prepare("SELECT * FROM schedule_state WHERE project_id = ? AND agent_id = ? AND name = ?") + .get(projectId, agentId, name); + return r ? mapRow(r as Record) : null; + } + + listByAgent(projectId: string, agentId: string): ScheduleStateRow[] { + const rows = this.db + .prepare("SELECT * FROM schedule_state WHERE project_id = ? AND agent_id = ? ORDER BY name") + .all(projectId, agentId); + return rows.map((r) => mapRow(r as Record)); + } + + /** + * Register or sync a task's runtime state, returning the synced row plus a `fresh` + * flag: + * - Insert if it doesn't exist (creator is only persisted at this point; a hand-edited + * file gets registered via reconciliation, with creator falling back to the Project + * owner); + * - A change to `start_at` resets the trigger state (a new task instance); + * - Otherwise, a change to the file fingerprint only clears the disabled flag. + * `fresh` = this call was an insert or reset — the scheduler only establishes its + * "missed, don't backfill" baseline at this moment; afterward, last_slot being NULL + * only means "no scheduled time has been consumed yet", and it must fire normally once + * reached. + */ + registerOrSync(args: { + projectId: string; + agentId: string; + name: string; + startAtMs: number; + defHash: string; + creatorUserId: string | null; + }): { row: ScheduleStateRow; fresh: boolean } { + const existing = this.find(args.projectId, args.agentId, args.name); + let fresh = false; + if (!existing) { + this.db + .prepare( + `INSERT INTO schedule_state + (project_id, agent_id, name, creator_user_id, start_at_ms, def_hash) + VALUES (?, ?, ?, ?, ?, ?)`, + ) + .run( + args.projectId, + args.agentId, + args.name, + args.creatorUserId, + args.startAtMs, + args.defHash, + ); + fresh = true; + } else if (existing.startAtMs !== args.startAtMs) { + this.db + .prepare( + `UPDATE schedule_state + SET start_at_ms = ?, def_hash = ?, last_slot_ms = NULL, last_fired_at = NULL, + fired_once = 0, missed = 0, invalid_reason = NULL + WHERE project_id = ? AND agent_id = ? AND name = ?`, + ) + .run(args.startAtMs, args.defHash, args.projectId, args.agentId, args.name); + fresh = true; + } else if (existing.defHash !== args.defHash) { + this.db + .prepare( + `UPDATE schedule_state SET def_hash = ?, invalid_reason = NULL + WHERE project_id = ? AND agent_id = ? AND name = ?`, + ) + .run(args.defHash, args.projectId, args.agentId, args.name); + } + const row = this.find(args.projectId, args.agentId, args.name); + if (!row) throw new Error("schedule_state 登记后读取失败"); + return { row, fresh }; + } + + /** Advance the consumed scheduled time (advances whether triggered or skipped; restarts don't re-trigger). */ + markSlot(projectId: string, agentId: string, name: string, slotMs: number): void { + this.db + .prepare( + `UPDATE schedule_state SET last_slot_ms = ? + WHERE project_id = ? AND agent_id = ? AND name = ?`, + ) + .run(slotMs, projectId, agentId, name); + } + + /** Record an actual send (also sets fired_once for a one-shot task). */ + markFired( + projectId: string, + agentId: string, + name: string, + firedAt: string, + oneShot: boolean, + ): void { + this.db + .prepare( + `UPDATE schedule_state SET last_fired_at = ?, fired_once = CASE WHEN ? THEN 1 ELSE fired_once END + WHERE project_id = ? AND agent_id = ? AND name = ?`, + ) + .run(firedAt, oneShot ? 1 : 0, projectId, agentId, name); + } + + /** Missed marker for a one-shot task (the scheduled time had already passed at startup/registration reconciliation; missed means not backfilled). */ + markMissed(projectId: string, agentId: string, name: string): void { + this.db + .prepare( + `UPDATE schedule_state SET missed = 1 + WHERE project_id = ? AND agent_id = ? AND name = ?`, + ) + .run(projectId, agentId, name); + } + + /** Mark as disabled (e.g. the bound Session was deleted); cleared via registerOrSync after the file is modified. */ + markInvalid(projectId: string, agentId: string, name: string, reason: string): void { + this.db + .prepare( + `UPDATE schedule_state SET invalid_reason = ? + WHERE project_id = ? AND agent_id = ? AND name = ?`, + ) + .run(reason, projectId, agentId, name); + } + + /** Deleting the file removes the task: clears its runtime state. */ + delete(projectId: string, agentId: string, name: string): void { + this.db + .prepare("DELETE FROM schedule_state WHERE project_id = ? AND agent_id = ? AND name = ?") + .run(projectId, agentId, name); + } + + /** Reconciliation cleanup: deletes state rows under this Agent that aren't in the current file list, returning the removed names. */ + deleteMissing(projectId: string, agentId: string, presentNames: string[]): string[] { + const rows = this.listByAgent(projectId, agentId); + const present = new Set(presentNames); + const removed: string[] = []; + for (const row of rows) { + if (!present.has(row.name)) { + this.delete(projectId, agentId, row.name); + removed.push(row.name); + } + } + return removed; + } + + deleteByProject(projectId: string): void { + this.db.prepare("DELETE FROM schedule_state WHERE project_id = ?").run(projectId); + } +} diff --git a/packages/server/src/db/repos/sessions.ts b/packages/server/src/db/repos/sessions.ts new file mode 100644 index 0000000..6980a41 --- /dev/null +++ b/packages/server/src/db/repos/sessions.ts @@ -0,0 +1,154 @@ +/** + * sessions table repo: + * Session index, approval mode, and auto-generated title; Session-level routes use this to look up project ownership. + */ +import type { DatabaseSync } from "node:sqlite"; +import type { ApprovalMode } from "../../api/types.js"; + +export interface SessionRow { + sessionId: string; + projectId: string; + agentId: string; + /** Provider group of the session's model (pairs with `modelId` to form the model reference). */ + provider: string; + /** Upstream model_id of the session's model (sent as-is to AgentHub; never concatenated). */ + modelId: string; + workspace: string; + approvalMode: ApprovalMode; + /** Auto-generated session title; NULL = not yet generated (frontend shows "New Conversation"). */ + title: string | null; + /** Archive timestamp, ISO; NULL = not archived (omitting on insert defaults to NULL). */ + archivedAt?: string | null; + /** Session origin: NULL = user-created; schedule = triggered by a Schedule; subagent = registered as a subagent session. */ + source?: "schedule" | "subagent" | null; + createdAt: string; +} + +function mapRow(r: Record): SessionRow { + return { + sessionId: r.session_id as string, + projectId: r.project_id as string, + agentId: r.agent_id as string, + provider: r.provider as string, + modelId: r.model_id as string, + workspace: r.workspace as string, + approvalMode: r.approval_mode as ApprovalMode, + title: (r.title as string | null) ?? null, + archivedAt: (r.archived_at as string | null) ?? null, + source: (r.source as "schedule" | "subagent" | null) ?? null, + createdAt: r.created_at as string, + }; +} + +export class SessionsRepo { + constructor(private readonly db: DatabaseSync) {} + + insert(row: SessionRow): void { + this.db + .prepare( + `INSERT INTO sessions (session_id, project_id, agent_id, provider, model_id, workspace, approval_mode, title, source, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + ) + .run( + row.sessionId, + row.projectId, + row.agentId, + row.provider, + row.modelId, + row.workspace, + row.approvalMode, + row.title, + row.source ?? null, + row.createdAt, + ); + } + + /** Idempotent insert: used when Trace directory discovery backfills a row (concurrent listing discovering the same Session no longer triggers a UNIQUE violation). */ + insertOrIgnore(row: SessionRow): void { + this.db + .prepare( + `INSERT OR IGNORE INTO sessions (session_id, project_id, agent_id, provider, model_id, workspace, approval_mode, title, source, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + ) + .run( + row.sessionId, + row.projectId, + row.agentId, + row.provider, + row.modelId, + row.workspace, + row.approvalMode, + row.title, + row.source ?? null, + row.createdAt, + ); + } + + findById(sessionId: string): SessionRow | null { + const r = this.db.prepare("SELECT * FROM sessions WHERE session_id = ?").get(sessionId); + return r ? mapRow(r) : null; + } + + listByAgent(projectId: string, agentId: string): SessionRow[] { + const rows = this.db + .prepare("SELECT * FROM sessions WHERE project_id = ? AND agent_id = ?") + .all(projectId, agentId); + return rows.map(mapRow); + } + + listByProject(projectId: string): SessionRow[] { + const rows = this.db.prepare("SELECT * FROM sessions WHERE project_id = ?").all(projectId); + return rows.map(mapRow); + } + + updateApprovalMode(sessionId: string, mode: ApprovalMode): void { + this.db + .prepare("UPDATE sessions SET approval_mode = ? WHERE session_id = ?") + .run(mode, sessionId); + } + + updateTitle(sessionId: string, title: string): void { + this.db.prepare("UPDATE sessions SET title = ? WHERE session_id = ?").run(title, sessionId); + } + + /** + * Writes only if the title is still NULL. Subagent-session registration and Trace + * directory discovery backfill can race on the same row, and both are insert-only: + * whichever inserts first determines the title. Discovery backfill can only supply NULL, + * so this method fills the title back in without overwriting an existing one (including + * a user rename or an already-generated title). + */ + updateTitleIfNull(sessionId: string, title: string): void { + this.db + .prepare("UPDATE sessions SET title = ? WHERE session_id = ? AND title IS NULL") + .run(title, sessionId); + } + + /** Archive / unarchive (archivedAt = ISO or NULL). */ + setArchived(sessionId: string, archivedAt: string | null): void { + this.db + .prepare("UPDATE sessions SET archived_at = ? WHERE session_id = ?") + .run(archivedAt, sessionId); + } + + /** Self-healing: after rebuilding a broken Session with no Trace, update the primary key to the new id. */ + replaceId(oldSessionId: string, newSessionId: string): void { + this.db + .prepare("UPDATE sessions SET session_id = ? WHERE session_id = ?") + .run(newSessionId, oldSessionId); + } + + deleteByAgent(projectId: string, agentId: string): void { + this.db + .prepare("DELETE FROM sessions WHERE project_id = ? AND agent_id = ?") + .run(projectId, agentId); + } + + deleteByProject(projectId: string): void { + this.db.prepare("DELETE FROM sessions WHERE project_id = ?").run(projectId); + } + + deleteById(sessionId: string): void { + this.db.prepare("DELETE FROM sessions WHERE session_id = ?").run(sessionId); + } +} diff --git a/packages/server/src/db/repos/ui-prefs.ts b/packages/server/src/db/repos/ui-prefs.ts new file mode 100644 index 0000000..984d5bc --- /dev/null +++ b/packages/server/src/db/repos/ui-prefs.ts @@ -0,0 +1,23 @@ +/** + * ui_prefs table repo (UI preferences): free-form JSON storage. + */ +import type { DatabaseSync } from "node:sqlite"; + +export class UiPrefsRepo { + constructor(private readonly db: DatabaseSync) {} + + /** Returns the raw JSON string; null if never set. */ + get(userId: string): string | null { + const r = this.db.prepare("SELECT prefs_json FROM ui_prefs WHERE user_id = ?").get(userId); + return r ? (r.prefs_json as string) : null; + } + + set(userId: string, prefsJson: string): void { + this.db + .prepare( + `INSERT INTO ui_prefs (user_id, prefs_json) VALUES (?, ?) + ON CONFLICT(user_id) DO UPDATE SET prefs_json = excluded.prefs_json`, + ) + .run(userId, prefsJson); + } +} diff --git a/packages/server/src/db/repos/usage.ts b/packages/server/src/db/repos/usage.ts new file mode 100644 index 0000000..9d853f9 --- /dev/null +++ b/packages/server/src/db/repos/usage.ts @@ -0,0 +1,247 @@ +/** + * usage_records table repo: + * one row per token_usage (per-request bucket). Stores Token counts only, not cost — + * cost is computed on the fly by usage-service against current pricing at query time, + * so every aggregation is broken down by the `(provider, model_id)` pair and returns + * raw Token sums (a model_id shared across providers is aggregated separately; never concatenated). + */ +import type { DatabaseSync } from "node:sqlite"; +import type { UsageGroupBy } from "../../api/types.js"; + +export interface UsageRecordInsert { + ts: string; + date: string; + projectId: string; + agentId: string; + sessionId: string; + originSessionId: string | null; + /** Provider group (pairs with modelId to form the attribution key). */ + provider: string; + /** Upstream model id (pairs with provider). */ + modelId: string; + cacheRead: number; + cacheWrite: number; + output: number; + total: number; + /** Request outcome; defaults to completed (success, carries tokens). Failed requests are stored with 0 tokens + status, for success-rate calculations. */ + status?: string; +} + +/** Generic filter: date range + agent / model dimensions (cost center top bar switches by agent/model). */ +export interface UsageFilter { + from?: string; + to?: string; + agentId?: string; + /** Provider filter paired with modelId (the frontend dropdown always sends them together). */ + provider?: string; + modelId?: string; +} + +/** + * Raw request success-rate counts for a single Model (paired reference). + * `total` is **the success-rate denominator**: all requests minus aborted — the user + * clicking "stop" is not a model failure, and counting it would make the success rate + * drop every time stop is pressed. `aborted` is counted separately for display. + */ +export interface UsageStatusCount { + provider: string; + modelId: string; + completed: number; + total: number; + aborted: number; + failed: number; + timeout: number; + malformed: number; +} + +/** Raw Token sums for a single Model (paired reference) — the smallest unit for cost conversion. */ +export interface UsageModelSums { + provider: string; + modelId: string; + cacheRead: number; + cacheWrite: number; + output: number; + total: number; + requests: number; +} + +/** Raw Token sums by group key x Model. */ +export interface UsageGroupModelSums extends UsageModelSums { + key: string; +} + +/** groupBy dimension -> column name allowlist (prevents injection; only these four columns can be group keys). */ +const GROUP_COLUMNS: Record = { + date: "date", + agent: "agent_id", + model: "model_id", + session: "session_id", +}; + +const SUM_COLUMNS = `COALESCE(SUM(cache_read), 0) AS cache_read, + COALESCE(SUM(cache_write), 0) AS cache_write, + COALESCE(SUM(output), 0) AS output, + COALESCE(SUM(total), 0) AS total, + COUNT(*) AS requests`; + +function toSums(r: Record): UsageModelSums { + return { + provider: r.provider as string, + modelId: r.model_id as string, + cacheRead: r.cache_read as number, + cacheWrite: r.cache_write as number, + output: r.output as number, + total: r.total as number, + requests: r.requests as number, + }; +} + +export class UsageRepo { + constructor(private readonly db: DatabaseSync) {} + + insert(r: UsageRecordInsert): void { + this.db + .prepare( + `INSERT INTO usage_records + (ts, date, project_id, agent_id, session_id, origin_session_id, provider, model_id, + cache_read, cache_write, output, total, status) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + ) + .run( + r.ts, + r.date, + r.projectId, + r.agentId, + r.sessionId, + r.originSessionId, + r.provider, + r.modelId, + r.cacheRead, + r.cacheWrite, + r.output, + r.total, + r.status ?? "completed", + ); + } + + /** WHERE fragment (project + optional date/agent/model) plus named params. */ + private conds( + projectId: string, + f: UsageFilter, + ): { where: string; params: Record } { + const conds = ["project_id = :pid"]; + const params: Record = { pid: projectId }; + if (f.from !== undefined) { + conds.push("date >= :from"); + params.from = f.from; + } + if (f.to !== undefined) { + conds.push("date <= :to"); + params.to = f.to; + } + if (f.agentId !== undefined) { + conds.push("agent_id = :agentId"); + params.agentId = f.agentId; + } + if (f.provider !== undefined) { + conds.push("provider = :provider"); + params.provider = f.provider; + } + if (f.modelId !== undefined) { + conds.push("model_id = :modelId"); + params.modelId = f.modelId; + } + return { where: conds.join(" AND "), params }; + } + + /** Sums (broken down by paired reference): date range + optional agent/model filter. */ + bucketByModel(projectId: string, f: UsageFilter = {}): UsageModelSums[] { + const { where, params } = this.conds(projectId, f); + const rows = this.db + .prepare( + `SELECT provider, model_id, ${SUM_COLUMNS} + FROM usage_records WHERE ${where} + GROUP BY provider, model_id`, + ) + .all(params); + return rows.map(toSums); + } + + /** Grouped aggregation (group key x paired reference breakdown): date range + optional agent/model filter. */ + groupsByModel( + projectId: string, + groupBy: UsageGroupBy, + f: UsageFilter = {}, + ): UsageGroupModelSums[] { + const col = GROUP_COLUMNS[groupBy]; + const { where, params } = this.conds(projectId, f); + const rows = this.db + .prepare( + `SELECT ${col} AS key, provider, model_id, ${SUM_COLUMNS} + FROM usage_records WHERE ${where} + GROUP BY ${col}, provider, model_id`, + ) + .all(params); + return rows.map((r) => ({ key: r.key as string, ...toSums(r) })); + } + + /** + * Raw success-rate counts per Model (paired reference) (completed / non-aborted requests): + * powers the cost center's "Model Success Rate" chart. The denominator excludes aborted + * (user-initiated interruption); failure breakdowns (failed / timeout / malformed) are + * also returned for hover display. Unknown statuses aren't broken out but still count + * toward the denominator (conservative: anything non-completed counts as a failure). + */ + statusByModel(projectId: string, f: UsageFilter = {}): UsageStatusCount[] { + const { where, params } = this.conds(projectId, f); + const count = (status: string) => + `COALESCE(SUM(CASE WHEN status = '${status}' THEN 1 ELSE 0 END), 0) AS ${status}`; + const rows = this.db + .prepare( + `SELECT provider, model_id, + ${count("completed")}, + ${count("aborted")}, + ${count("failed")}, + ${count("timeout")}, + ${count("malformed")}, + COALESCE(SUM(CASE WHEN status <> 'aborted' THEN 1 ELSE 0 END), 0) AS total + FROM usage_records WHERE ${where} GROUP BY provider, model_id`, + ) + .all(params); + return rows.map((r) => ({ + provider: r.provider as string, + modelId: r.model_id as string, + completed: r.completed as number, + total: r.total as number, + aborted: r.aborted as number, + failed: r.failed as number, + timeout: r.timeout as number, + malformed: r.malformed as number, + })); + } + + /** Distinct agent_id values seen for this Project (for filter dropdowns). */ + distinctAgentIds(projectId: string): string[] { + const rows = this.db + .prepare( + "SELECT DISTINCT agent_id AS v FROM usage_records WHERE project_id = ? ORDER BY agent_id", + ) + .all(projectId); + return rows.map((r) => r.v as string); + } + + /** Distinct Model paired references seen for this Project (for filter dropdowns). */ + distinctModels(projectId: string): Array<{ provider: string; modelId: string }> { + const rows = this.db + .prepare( + `SELECT DISTINCT provider, model_id FROM usage_records + WHERE project_id = ? ORDER BY provider, model_id`, + ) + .all(projectId); + return rows.map((r) => ({ provider: r.provider as string, modelId: r.model_id as string })); + } + + deleteByProject(projectId: string): void { + this.db.prepare("DELETE FROM usage_records WHERE project_id = ?").run(projectId); + } +} diff --git a/packages/server/src/db/repos/users.ts b/packages/server/src/db/repos/users.ts new file mode 100644 index 0000000..3326303 --- /dev/null +++ b/packages/server/src/db/repos/users.ts @@ -0,0 +1,70 @@ +/** + * users table repo: pure SQL wrapper, no business rules. + * user_id is the login name (a semantic id, specified at creation, immutable). + */ +import type { DatabaseSync } from "node:sqlite"; + +export interface UserRow { + userId: string; + passwordHash: string; + isAdmin: boolean; + /** Still using the initial password (seeded / set by an admin); cleared to 0 once the user changes it. */ + passwordIsInitial: boolean; + createdAt: string; +} + +function mapRow(r: Record): UserRow { + return { + userId: r.user_id as string, + passwordHash: r.password_hash as string, + isAdmin: (r.is_admin as number) === 1, + passwordIsInitial: (r.password_is_initial as number) === 1, + createdAt: r.created_at as string, + }; +} + +export class UsersRepo { + constructor(private readonly db: DatabaseSync) {} + + insert(row: UserRow): void { + this.db + .prepare( + "INSERT INTO users (user_id, password_hash, is_admin, password_is_initial, created_at) VALUES (?, ?, ?, ?, ?)", + ) + .run( + row.userId, + row.passwordHash, + row.isAdmin ? 1 : 0, + row.passwordIsInitial ? 1 : 0, + row.createdAt, + ); + } + + findById(userId: string): UserRow | null { + const r = this.db.prepare("SELECT * FROM users WHERE user_id = ?").get(userId); + return r ? mapRow(r) : null; + } + + /** All users (for the admin user backend), ordered by creation time ascending. */ + list(): UserRow[] { + const rows = this.db.prepare("SELECT * FROM users ORDER BY created_at ASC, user_id ASC").all(); + return rows.map(mapRow); + } + + count(): number { + const r = this.db.prepare("SELECT COUNT(*) AS n FROM users").get(); + return (r?.n as number) ?? 0; + } + + /** Update the password hash; isInitial marks whether the password was set by someone else (seed / admin). */ + updatePassword(userId: string, passwordHash: string, isInitial: boolean): void { + this.db + .prepare("UPDATE users SET password_hash = ?, password_is_initial = ? WHERE user_id = ?") + .run(passwordHash, isInitial ? 1 : 0, userId); + } + + /** Used by admin user deletion and account-creation compensation paths (owned Projects must be cleaned up first). */ + delete(userId: string): void { + this.db.prepare("DELETE FROM users WHERE user_id = ?").run(userId); + } +} diff --git a/packages/server/src/db/schema.ts b/packages/server/src/db/schema.ts new file mode 100644 index 0000000..1ecb89a --- /dev/null +++ b/packages/server/src/db/schema.ts @@ -0,0 +1,104 @@ +/** + * SQLite table-creation SQL. + * + * SQLite stores only indexes and aggregates: users / login sessions / Project authorization / + * Agent & Session indexes / usage summaries / error records / UI preferences. Agent State, + * Trace, and Workspace still follow the local directory-based storage rules. + * Product not yet released: no migration branches — everything is CREATE IF NOT EXISTS, formed once. + */ + +export const SCHEMA_SQL = ` +CREATE TABLE IF NOT EXISTS users ( + user_id TEXT PRIMARY KEY, -- 语义 id 即登录名:^[a-z][a-z0-9_-]{1,31}$ + password_hash TEXT NOT NULL, -- scrypt$N$r$p$salt$hash(base64) + is_admin INTEGER NOT NULL DEFAULT 0, -- 内置 admin(启动时种子)为 1 + password_is_initial INTEGER NOT NULL DEFAULT 0, -- 1=初始密码(种子/管理员设置);本人改密后清 0 + created_at TEXT NOT NULL +); +CREATE TABLE IF NOT EXISTS auth_sessions ( + token_hash TEXT PRIMARY KEY, -- sha256(token) hex;cookie 只存原始 token + user_id TEXT NOT NULL REFERENCES users(user_id) ON DELETE CASCADE, + created_at TEXT NOT NULL, + expires_at TEXT NOT NULL -- 7 天滑动续期(剩余 <6 天则续满) +); +CREATE TABLE IF NOT EXISTS projects ( + project_id TEXT PRIMARY KEY, -- 目录名即 id;显示名在 project_config.toml + owner_user_id TEXT NOT NULL REFERENCES users(user_id), + created_at TEXT NOT NULL +); +CREATE TABLE IF NOT EXISTS project_members ( -- 仅 member 授权关系;owner 不入此表 + project_id TEXT NOT NULL REFERENCES projects(project_id) ON DELETE CASCADE, + user_id TEXT NOT NULL REFERENCES users(user_id) ON DELETE CASCADE, + created_at TEXT NOT NULL, + PRIMARY KEY (project_id, user_id) +); +CREATE TABLE IF NOT EXISTS agents ( -- 索引;name/description 在 system_config.yaml + project_id TEXT NOT NULL, + agent_id TEXT NOT NULL, + created_at TEXT NOT NULL, + PRIMARY KEY (project_id, agent_id) +); +CREATE TABLE IF NOT EXISTS sessions ( + session_id TEXT PRIMARY KEY, + project_id TEXT NOT NULL, + agent_id TEXT NOT NULL, + provider TEXT NOT NULL, -- 会话模型的厂商分组(与 model_id 成对构成模型引用) + model_id TEXT NOT NULL, -- 上游模型 id(原样发给 AgentHub;禁止 / 拼接) + workspace TEXT NOT NULL, + approval_mode TEXT NOT NULL DEFAULT 'allow-all', -- allow-all|deny-all|read-only|always-ask + title TEXT, -- 首次对话后由模型自动生成;NULL=未生成(前端显示「新对话」) + archived_at TEXT, -- 归档时刻;NULL=未归档(默认展示;归档后收进「已归档」) + source TEXT, -- 会话来源:NULL=用户创建 | schedule(定时任务)| subagent(子会话) + created_at TEXT NOT NULL +); +CREATE TABLE IF NOT EXISTS usage_records ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + ts TEXT NOT NULL, + date TEXT NOT NULL, -- 本地日期 yyyy-mm-dd(聚合键,与 Trace 日期目录同口径) + project_id TEXT NOT NULL, + agent_id TEXT NOT NULL, + session_id TEXT NOT NULL, -- 顶层 Session(子会话消耗计入所属主 Session) + origin_session_id TEXT, -- 直接来源子 Session(origin 链末项);NULL=主会话 + provider TEXT NOT NULL, -- 厂商分组:与 model_id 成对构成归因键,聚合一律 GROUP BY provider, model_id + model_id TEXT NOT NULL, -- 上游模型 id(与 provider 成对;同名 model_id 跨厂商分开聚合) + cache_read INTEGER NOT NULL, + cache_write INTEGER NOT NULL, + output INTEGER NOT NULL, + total INTEGER NOT NULL, -- 取 token_usage.request(每 Request 一条) + status TEXT NOT NULL DEFAULT 'completed' -- 请求结局:completed=成功(含 token);其余=失败(0 token,供成功率) +); -- 成本不落库:查询时按当前 pricing 实时折算 +CREATE INDEX IF NOT EXISTS idx_usage_project_date ON usage_records(project_id, date); +CREATE INDEX IF NOT EXISTS idx_usage_session ON usage_records(session_id); +CREATE TABLE IF NOT EXISTS error_records ( -- 服务端异常捕获(统计中心「异常」) + id INTEGER PRIMARY KEY AUTOINCREMENT, + ts TEXT NOT NULL, + date TEXT NOT NULL, -- 本地日期 yyyy-mm-dd(与 usage_records 同口径) + project_id TEXT, -- 可空:登录/注册、进程级异常没有 Project 上下文 + agent_id TEXT, + session_id TEXT, + source TEXT NOT NULL, -- http | session | usage | title | subagent | process | llm | environment | schedule + kind TEXT NOT NULL, -- expected(HttpError,业务 4xx)| unexpected(500/运行时) + code TEXT NOT NULL, -- HttpError.code / internal / session_run_failed / ... + status INTEGER, -- HTTP 状态码;非 HTTP 来源为 NULL + message TEXT NOT NULL -- 截断 500 字符(不存堆栈:堆栈只进日志) +); +CREATE INDEX IF NOT EXISTS idx_error_project_date ON error_records(project_id, date); +CREATE TABLE IF NOT EXISTS schedule_state ( -- 定时任务运行状态(文件是声明式意图,系统不写回) + project_id TEXT NOT NULL, + agent_id TEXT NOT NULL, + name TEXT NOT NULL, -- 文件名(去 .toml)即标识 + creator_user_id TEXT, -- 创建者(API 创建时记;手编文件对账登记回退 Project owner) + start_at_ms INTEGER NOT NULL, -- 定义身份:start_at 变更视为新任务实例,重置触发状态 + def_hash TEXT NOT NULL, -- 文件内容指纹:变更即清除失效标记(文件修改后重新生效) + last_slot_ms INTEGER, -- 已消化的最近应触发时刻(触发或跳过都推进;重启不重复触发) + last_fired_at TEXT, -- 最近实际发送时刻(展示用) + fired_once INTEGER NOT NULL DEFAULT 0, -- 一次性任务已触发 + missed INTEGER NOT NULL DEFAULT 0, -- 一次性任务已错过(启动/登记对账标记,错过不补) + invalid_reason TEXT, -- 失效原因(如绑定 Session 已删除);NULL 即正常 + PRIMARY KEY (project_id, agent_id, name) +); +CREATE TABLE IF NOT EXISTS ui_prefs ( + user_id TEXT PRIMARY KEY REFERENCES users(user_id) ON DELETE CASCADE, + prefs_json TEXT NOT NULL -- {theme?, lastProjectId?, ...} 自由 JSON +); +`; diff --git a/packages/server/src/http/errors.ts b/packages/server/src/http/errors.ts new file mode 100644 index 0000000..a9da343 --- /dev/null +++ b/packages/server/src/http/errors.ts @@ -0,0 +1,55 @@ +/** + * Unified HTTP error: `{error: {code, message}}` response body, + * with message in Chinese. + * + * Routes and services express business errors via throw HttpError; app-level onError + * uniformly converges these into a JSON response, with unknown errors converged to 500 + * (never leaking internal details to the client). + */ +import type { Context } from "hono"; +import type { ErrorBody } from "../api/types.js"; + +export class HttpError extends Error { + constructor( + readonly status: number, + readonly code: string, + message: string, + ) { + super(message); + this.name = "HttpError"; + } +} + +export function errorBody(code: string, message: string): ErrorBody { + return { error: { code, message } }; +} + +/** + * Model missing a credential: the provider SDK throws this at **client-construction + * time** (hit by both creating a Session and resuming a Session); the original message + * is full of environment variable names, meaningless to a user — uniformly replaced + * with a single actionable sentence. The frontend produces localized text by code + * (message is only a fallback); see web's lib/api-error.ts. + */ +export function isMissingCredential(err: unknown): boolean { + const message = err instanceof Error ? err.message : String(err); + return /missing credentials|api[_ ]?key/i.test(message); +} + +export function modelCredentialMissing(modelId: string): HttpError { + return new HttpError( + 400, + "model_credential_missing", + `模型 ${modelId} 还没有可用的 API key,请先在「模型」页为它配置。`, + ); +} + +/** app.onError handler: maps HttpError through as-is; everything else is logged and converged to 500. */ +export function handleError(err: Error, c: Context): Response { + if (err instanceof HttpError) { + return c.json(errorBody(err.code, err.message), err.status as 400); + } + // Unknown error: print the stack for diagnosis, but never expose details externally. + console.error(`[server] 未处理异常: ${err.stack ?? err.message}`); + return c.json(errorBody("internal", "服务器内部错误。"), 500); +} diff --git a/packages/server/src/http/routes/admin.ts b/packages/server/src/http/routes/admin.ts new file mode 100644 index 0000000..68ef49b --- /dev/null +++ b/packages/server/src/http/routes/admin.ts @@ -0,0 +1,47 @@ +/** + * Admin user-backend routes: only the built-in admin can use these (403 for non-admins). + * GET|POST /api/admin/users, POST /api/admin/users/:userId/password, DELETE /api/admin/users/:userId. + */ +import { Hono } from "hono"; +import type { AdminUserCreateResponse, AdminUsersResponse } from "../../api/types.js"; +import { HttpError } from "../errors.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { pathParam, readJson, requireString } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +export function adminUsersRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.use("*", async (c, next) => { + if (!c.var.user.isAdmin) { + throw new HttpError(403, "admin_required", "该操作仅管理员可执行。"); + } + await next(); + }); + + app.get("/", (c) => { + return c.json({ users: deps.adminService.listUsers() } satisfies AdminUsersResponse); + }); + + app.post("/", async (c) => { + const body = await readJson(c); + const userId = requireString(body, "userId", { label: "userId" }); + const password = requireString(body, "password", { label: "password" }); + const user = await deps.adminService.createUser(userId, password); + return c.json({ user } satisfies AdminUserCreateResponse, 201); + }); + + app.post("/:userId/password", async (c) => { + const body = await readJson(c); + const password = requireString(body, "password", { label: "password" }); + await deps.adminService.resetPassword(pathParam(c, "userId"), password); + return c.body(null, 204); + }); + + app.delete("/:userId", async (c) => { + await deps.adminService.deleteUser(pathParam(c, "userId")); + return c.body(null, 204); + }); + + return app; +} diff --git a/packages/server/src/http/routes/agent-config.ts b/packages/server/src/http/routes/agent-config.ts new file mode 100644 index 0000000..311dfe5 --- /dev/null +++ b/packages/server/src/http/routes/agent-config.ts @@ -0,0 +1,50 @@ +/** + * Agent config routes (reads/writes system_config.yaml and AGENTS.md): + * GET|PUT /api/projects/:p/agents/:a/config. Members can read and write (unrestricted). + */ +import { Hono } from "hono"; +import type { AgentConfigResponse, AgentConfigUpdateRequest } from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { badRequest, optionalString, readJson, requireValidId } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +export function agentConfigRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + // Id validation happens before any path construction (FD-4: prevents agentId path traversal for cross-Project privilege escalation). + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + const view = await deps.agentConfigService.getConfig(projectId, agentId); + return c.json({ + ...view, + activeSessionCount: deps.manager.activeCountForAgent(projectId, agentId), + } satisfies AgentConfigResponse); + }); + + app.put("/", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + const body = await readJson(c); + const req: AgentConfigUpdateRequest = {}; + const agentsMd = optionalString(body, "agentsMd", { label: "agentsMd" }); + if (agentsMd !== undefined) req.agentsMd = agentsMd; + if (body.config !== undefined) { + if (body.config === null || typeof body.config !== "object" || Array.isArray(body.config)) { + throw badRequest("config 必须是对象。"); + } + req.config = body.config as AgentConfigUpdateRequest["config"]; + } + // Fine-grained validation (numeric ranges / enums) is done inside agent-config-service. + await deps.agentConfigService.updateConfig(projectId, agentId, req); + const view = await deps.agentConfigService.getConfig(projectId, agentId); + return c.json({ + ...view, + activeSessionCount: deps.manager.activeCountForAgent(projectId, agentId), + } satisfies AgentConfigResponse); + }); + + return app; +} diff --git a/packages/server/src/http/routes/agent-traces.ts b/packages/server/src/http/routes/agent-traces.ts new file mode 100644 index 0000000..49a3cae --- /dev/null +++ b/packages/server/src/http/routes/agent-traces.ts @@ -0,0 +1,48 @@ +/** + * Agent-level Trace browsing routes: + * - GET /api/projects/:p/agents/:a/traces — drills down Agent -> date -> Session -> index (reverse order); + * - GET /api/projects/:p/agents/:a/traces/:sessionId/:index (including /analysis) — + * read-only Trace detail endpoints (FD-3): locate the Trace file directly by + * (projectId, agentId, sessionId), without depending on the sessions table for + * tracking — any entry visible in the directory tree (subagent child Sessions, + * CLI-created Sessions) can be opened and read; access is enforced by requireProjectAccess. + */ +import { Hono } from "hono"; +import type { AppEnv } from "../../auth/middleware.js"; +import { paginationQuery, positiveIntParam, requireValidId } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +export function agentTracesRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + // Id validation happens before any path construction (FD-4: prevents agentId path traversal for cross-Project privilege escalation). + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + return c.json(await deps.traceService.agentTraces(projectId, agentId)); + }); + + app.get("/:sessionId/:index", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + const sessionId = requireValidId(c, "sessionId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + const index = positiveIntParam(c, "index"); + const { offset, limit } = paginationQuery(c); + return c.json( + await deps.traceService.readEvents(projectId, agentId, sessionId, index, offset, limit), + ); + }); + + app.get("/:sessionId/:index/analysis", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + const sessionId = requireValidId(c, "sessionId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + const index = positiveIntParam(c, "index"); + return c.json(await deps.traceService.analyze(projectId, agentId, sessionId, index)); + }); + + return app; +} diff --git a/packages/server/src/http/routes/agent-transfer.ts b/packages/server/src/http/routes/agent-transfer.ts new file mode 100644 index 0000000..2f8a896 --- /dev/null +++ b/packages/server/src/http/routes/agent-transfer.ts @@ -0,0 +1,55 @@ +/** + * Agent State export/import routes: + * GET /api/projects/:p/agents/:a/export (any member; auto-packages if no snapshot exists, downloads tar.gz) + * POST /api/projects/:p/agents/:a/import (owner only; version conflicts require a confirm flag) + */ +import fs from "node:fs/promises"; +import { Hono } from "hono"; +import type { AgentImportResponse } from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import type { AppDeps } from "../../app.js"; +import { badRequest, readJson, requireString, requireValidId } from "../validate.js"; + +/** Import archive size cap: aligned with the global request body limit (stays within 20MB after base64). */ +const MAX_ARCHIVE_BYTES = 14 * 1024 * 1024; + +export function agentTransferRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/export", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + const { file, fileName } = await deps.snapshots.exportArchive(projectId, agentId); + const bytes = await fs.readFile(file); + return new Response(new Uint8Array(bytes), { + headers: { + "Content-Type": "application/gzip", + "Content-Disposition": `attachment; filename="${fileName}"`, + }, + }); + }); + + app.post("/import", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectOwner(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + const body = await readJson(c); + const dataBase64 = requireString(body, "dataBase64", { minLen: 1, maxLen: 20 * 1024 * 1024 }); + const confirm = body.confirm === true; + let archive: Buffer; + try { + archive = Buffer.from(dataBase64, "base64"); + } catch { + throw badRequest("dataBase64 不是合法的 base64。"); + } + if (archive.byteLength === 0) throw badRequest("导入包为空。"); + if (archive.byteLength > MAX_ARCHIVE_BYTES) throw badRequest("导入包超过 14MB 上限。"); + const { version } = await deps.snapshots.importArchive(projectId, agentId, archive, confirm); + const res: AgentImportResponse = { version }; + return c.json(res); + }); + + return app; +} diff --git a/packages/server/src/http/routes/agents.ts b/packages/server/src/http/routes/agents.ts new file mode 100644 index 0000000..2c03766 --- /dev/null +++ b/packages/server/src/http/routes/agents.ts @@ -0,0 +1,86 @@ +/** + * Agent routes: + * GET|POST /api/projects/:p/agents, DELETE /:agentId (owner only). + * The list is the union of DB entries and directory scan results, including active + * Session count, total Session count, and config last-modified time. + */ +import { Hono } from "hono"; +import type { AgentCreateResponse, AgentsResponse, AgentSummary } from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { settleWithin } from "../settle.js"; +import { optionalString, readJson, requireString, requireValidId } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +/** Window size in days for the card's activity sparkline (last 30 days, including today). */ +const ACTIVITY_DAYS = 30; + +export function agentsRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + // Defensive id validation (FD-4): don't rely on the implicit invariant that requireProjectAccess always runs before path construction. + const projectId = requireValidId(c, "projectId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + const items = await deps.agentService.listAgents(projectId); + const agents: AgentSummary[] = await Promise.all( + items.map(async (item) => { + const stats = await deps.sessionService.sessionStats( + projectId, + item.agentId, + ACTIVITY_DAYS, + ); + return { + ...item, + activeSessionCount: deps.manager.activeCountForAgent(projectId, item.agentId), + sessionCount: stats.sessionCount, + sessionActivity: stats.activity, + }; + }), + ); + return c.json({ agents } satisfies AgentsResponse); + }); + + app.post("/", async (c) => { + const projectId = requireValidId(c, "projectId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + const body = await readJson(c); + const agentId = requireString(body, "agentId", { label: "agentId" }); + const name = optionalString(body, "name", { minLen: 1, maxLen: 100, label: "name" }); + const description = optionalString(body, "description", { + maxLen: 2000, + label: "description", + }); + const item = await deps.agentService.createAgent(projectId, agentId, name, description); + const agent: AgentSummary = { + ...item, + activeSessionCount: 0, + sessionCount: 0, + sessionActivity: Array.from({ length: ACTIVITY_DAYS }, () => 0), + }; + return c.json({ agent } satisfies AgentCreateResponse, 201); + }); + + app.delete("/:agentId", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + // Deletion is a Project-level management operation: owner only. + deps.projectService.requireProjectOwner(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + // Mark as deleting and converge active runs (beginAgentDeletion): any new Task during + // this window gets 409, preventing the race where a new task recreates the directory + // and revives the Agent between abort and rm. Abort cleanup writes the Trace + // asynchronously; wait for it to finish before removing the directory, and clear the + // deleting flag once deletion completes (success or failure). + const runnings = deps.manager.beginAgentDeletion(projectId, agentId); + try { + await settleWithin(runnings, 5000); + await deps.agentService.deleteAgent(projectId, agentId); + deps.sessionsRepo.deleteByAgent(projectId, agentId); + } finally { + deps.manager.endAgentDeletion(projectId, agentId); + } + return c.body(null, 204); + }); + + return app; +} diff --git a/packages/server/src/http/routes/auth.ts b/packages/server/src/http/routes/auth.ts new file mode 100644 index 0000000..b795f29 --- /dev/null +++ b/packages/server/src/http/routes/auth.ts @@ -0,0 +1,46 @@ +/** + * Auth routes: POST /api/auth/login | logout. + * No self-registration: users are created by an admin in the user backend (/api/admin/users). + * Login issues a cookie session; logout deletes the server-side session and clears the cookie. + */ +import { Hono } from "hono"; +import { deleteCookie, getCookie, setCookie } from "hono/cookie"; +import type { AuthResponse } from "../../api/types.js"; +import { SESSION_COOKIE } from "../../auth/middleware.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { readJson, requireString } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +/** Session cookie attributes: HttpOnly, SameSite=Lax, 7 days. */ +function cookieOptions(c: { req: { header(name: string): string | undefined } }) { + return { + httpOnly: true, + sameSite: "Lax" as const, + path: "/", + maxAge: 7 * 24 * 60 * 60, + // Add Secure when the reverse proxy declares https. + ...(c.req.header("x-forwarded-proto") === "https" ? { secure: true } : {}), + }; +} + +export function authRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.post("/login", async (c) => { + const body = await readJson(c); + const userId = requireString(body, "userId", { label: "userId" }); + const password = requireString(body, "password", { label: "password" }); + const { user, token } = await deps.authService.login(userId, password); + setCookie(c, SESSION_COOKIE, token, cookieOptions(c)); + return c.json({ user } satisfies AuthResponse); + }); + + app.post("/logout", (c) => { + const token = getCookie(c, SESSION_COOKIE); + if (token) deps.authService.logout(token); + deleteCookie(c, SESSION_COOKIE, { path: "/" }); + return c.body(null, 204); + }); + + return app; +} diff --git a/packages/server/src/http/routes/benchmarks.ts b/packages/server/src/http/routes/benchmarks.ts new file mode 100644 index 0000000..f66a12d --- /dev/null +++ b/packages/server/src/http/routes/benchmarks.ts @@ -0,0 +1,24 @@ +/** + * Benchmark scoring routes: + * GET /api/projects/:p/agents/:a/benchmarks (any member, read-only) + * Returns the Agent's Benchmark list (title/description from benchmark_config.toml) + * along with the evaluations[] from scoreboard.yaml. + */ +import { Hono } from "hono"; +import type { AppEnv } from "../../auth/middleware.js"; +import type { AppDeps } from "../../app.js"; +import { requireValidId } from "../validate.js"; + +export function benchmarksRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + return c.json(await deps.benchmarks.list(projectId, agentId)); + }); + + return app; +} diff --git a/packages/server/src/http/routes/dirs.ts b/packages/server/src/http/routes/dirs.ts new file mode 100644 index 0000000..e94fa18 --- /dev/null +++ b/packages/server/src/http/routes/dirs.ts @@ -0,0 +1,69 @@ +/** + * Server directory browsing: + * GET /api/projects/:p/dirs?path=. + * + * Lets the user interactively pick a Workspace directory when creating a Session via + * advanced mode. Defaults to the home directory of the account running the service, and + * can be browsed all the way up to the root `/` — reachability is governed by OS file + * permissions; the server no longer restricts browsing to within the Project directory + * tree (same convention as workspace-guard). Lists subdirectories only, not files. + * + * `projectId` remains the authorization anchor: the caller must have access to that Project. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { Hono } from "hono"; +import type { DirListResponse } from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { HttpError } from "../errors.js"; +import { requireValidId } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +export function dirsRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + const projectId = requireValidId(c, "projectId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + + // Default starting point: home directory; an explicit path must be absolute (the frontend always sends back the realpath result). + const raw = c.req.query("path"); + const target = raw && raw.trim() ? raw.trim() : os.homedir(); + if (!path.isAbsolute(target)) { + throw new HttpError(400, "dir_not_absolute", "目录必须是绝对路径。"); + } + + let real: string; + try { + real = await fs.realpath(target); + } catch { + throw new HttpError(404, "dir_not_found", `目录不存在或不可访问:${target}。`); + } + const stat = await fs.stat(real); + if (!stat.isDirectory()) { + throw new HttpError(400, "not_a_dir", "不是目录。"); + } + + let dirents: import("node:fs").Dirent[] = []; + try { + dirents = await fs.readdir(real, { withFileTypes: true }); + } catch { + // No read permission: return an empty list instead of an error, so the user can still navigate back up. + dirents = []; + } + const entries = dirents + .filter((d) => d.isDirectory()) + .map((d) => ({ name: d.name, path: path.join(real, d.name) })) + .sort((a, b) => a.name.localeCompare(b.name)); + + const parent = path.dirname(real); + return c.json({ + path: real, + parent: parent === real ? null : parent, + entries, + } satisfies DirListResponse); + }); + + return app; +} diff --git a/packages/server/src/http/routes/events.ts b/packages/server/src/http/routes/events.ts new file mode 100644 index 0000000..7084d7c --- /dev/null +++ b/packages/server/src/http/routes/events.ts @@ -0,0 +1,24 @@ +/** + * User-level server event stream: GET /api/events (SSE user channel). + * Carries cross-Session notifications (reserved for automated tasks); sends a `hello` handshake event on connect. + */ +import { Hono } from "hono"; +import type { AppEnv } from "../../auth/middleware.js"; +import { sseEndpoint } from "../sse.js"; +import type { AppDeps } from "../../app.js"; + +/** The user channel's key in ChannelHub. */ +export function userChannelKey(userId: string): string { + return `user:${userId}`; +} + +export function eventsRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", (c) => { + const channel = deps.channels.get(userChannelKey(c.var.user.userId)); + return sseEndpoint(c, channel, { initialEvents: [{ type: "hello" }] }); + }); + + return app; +} diff --git a/packages/server/src/http/routes/me.ts b/packages/server/src/http/routes/me.ts new file mode 100644 index 0000000..2185c0b --- /dev/null +++ b/packages/server/src/http/routes/me.ts @@ -0,0 +1,65 @@ +/** + * Current-user routes: GET /api/me, PUT /api/me/password, GET|PUT /api/me/prefs. + * ui_prefs is free-form JSON (theme / lastProjectId / credentialGuideSeen, etc.): GET reads + * it whole, PUT shallow-merges (PATCH semantics) — several independent writers each write + * their own fields without clobbering each other. + */ +import { Hono } from "hono"; +import type { MeResponse, PrefsResponse, UiPrefs } from "../../api/types.js"; +import { toUserInfo } from "../../auth/service.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { readJson, requireString } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +export function meRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", (c) => { + return c.json({ user: toUserInfo(c.var.user) } satisfies MeResponse); + }); + + // Self-service password change (user settings): validates the old password; on success, the initial-password prompt disappears from GET /api/me. + app.put("/password", async (c) => { + const body = await readJson(c); + const oldPassword = requireString(body, "oldPassword", { label: "oldPassword" }); + const newPassword = requireString(body, "newPassword", { label: "newPassword" }); + await deps.authService.changePassword(c.var.user.userId, oldPassword, newPassword); + return c.body(null, 204); + }); + + app.get("/prefs", (c) => { + const raw = deps.prefsRepo.get(c.var.user.userId); + let prefs: UiPrefs = {}; + if (raw !== null) { + try { + prefs = JSON.parse(raw) as UiPrefs; + } catch { + prefs = {}; // Corrupted prefs fall back to an empty object + } + } + return c.json({ prefs } satisfies PrefsResponse); + }); + + // PATCH semantics: the request body is **shallow-merged** into existing prefs, not a + // full replace. prefs has several independent writers (lastProjectId / + // credentialGuideSeen, etc., each writing their own field); a full replace would wipe + // out each other's fields — e.g. writing lastProjectId when switching Projects would + // clear credentialGuideSeen, breaking the "show onboarding once ever" guarantee. + app.put("/prefs", async (c) => { + const body = await readJson(c); + const raw = deps.prefsRepo.get(c.var.user.userId); + let current: UiPrefs = {}; + if (raw !== null) { + try { + current = JSON.parse(raw) as UiPrefs; + } catch { + current = {}; // Corrupted prefs fall back to an empty object (consistent with GET). + } + } + const merged = { ...current, ...(body as UiPrefs) }; + deps.prefsRepo.set(c.var.user.userId, JSON.stringify(merged)); + return c.json({ prefs: merged } satisfies PrefsResponse); + }); + + return app; +} diff --git a/packages/server/src/http/routes/members.ts b/packages/server/src/http/routes/members.ts new file mode 100644 index 0000000..f747781 --- /dev/null +++ b/packages/server/src/http/routes/members.ts @@ -0,0 +1,45 @@ +/** + * Member authorization routes: + * GET|POST /api/projects/:p/members, DELETE /api/projects/:p/members/:userId. + * Reading requires access; adding/removing is owner-only (validated inside the service). + */ +import { Hono } from "hono"; +import type { MemberAddResponse, MembersResponse } from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { pathParam, readJson, requireString, requireValidId } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +export function membersRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", (c) => { + // Defensive id validation (FD-4). + const members = deps.projectService.listMembers( + c.var.user.userId, + requireValidId(c, "projectId"), + ); + return c.json({ members } satisfies MembersResponse); + }); + + app.post("/", async (c) => { + const body = await readJson(c); + const userId = requireString(body, "userId", { label: "userId" }); + const member = deps.projectService.addMember( + c.var.user.userId, + requireValidId(c, "projectId"), + userId, + ); + return c.json({ member } satisfies MemberAddResponse, 201); + }); + + app.delete("/:userId", (c) => { + deps.projectService.removeMember( + c.var.user.userId, + requireValidId(c, "projectId"), + pathParam(c, "userId"), + ); + return c.body(null, 204); + }); + + return app; +} diff --git a/packages/server/src/http/routes/models.ts b/packages/server/src/http/routes/models.ts new file mode 100644 index 0000000..3d0a82e --- /dev/null +++ b/packages/server/src/http/routes/models.ts @@ -0,0 +1,178 @@ +/** + * Model & credential config routes: + * GET|PUT /api/projects/:p/models, POST /api/projects/:p/models/test (the model reference + * `(provider, modelId)` is sent as a pair in the request body, avoiding URL-encoding + * issues). Any member can read (api_key is masked); only the owner can modify or test. + */ +import { Hono } from "hono"; +import type { + ModelRefDto, + ModelsUpdateRequest, + ModelTestRequest, + ModelUpdateEntry, +} from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { badRequest, readJson, requireString, requireValidId } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +/** Validate a paired reference object ({ provider, modelId }); shape mismatch throws 400. */ +function parseRef(value: unknown, label: string): ModelRefDto { + if (value === null || typeof value !== "object" || Array.isArray(value)) { + throw badRequest(`${label} 必须是 { provider, modelId } 对象。`); + } + const r = value as Record; + return { + provider: requireString(r, "provider", { minLen: 1, maxLen: 64, label: `${label}.provider` }), + modelId: requireString(r, "modelId", { minLen: 1, maxLen: 200, label: `${label}.modelId` }), + }; +} + +/** Validate the PUT request body and shape it into a ModelsUpdateRequest (rejects any shape errors). */ +function parseModelsUpdate(body: Record): ModelsUpdateRequest { + if (!Array.isArray(body.models)) throw badRequest("models 必须是数组。"); + const models: ModelUpdateEntry[] = body.models.map((item, i) => { + if (item === null || typeof item !== "object" || Array.isArray(item)) { + throw badRequest(`models[${i}] 必须是对象。`); + } + const m = item as Record; + const entry: ModelUpdateEntry = { + provider: requireString(m, "provider", { + minLen: 1, + maxLen: 64, + label: `models[${i}].provider`, + }), + modelId: requireString(m, "modelId", { + minLen: 1, + maxLen: 200, + label: `models[${i}].modelId`, + }), + }; + if (m.displayName !== undefined) { + if (typeof m.displayName !== "string" || m.displayName.length > 100) { + throw badRequest(`models[${i}].displayName 必须是长度不超过 100 的字符串。`); + } + if (m.displayName) entry.displayName = m.displayName; + } + // A key change (either the provider group or the upstream id) goes through renamedFrom's paired old reference; unknown fields are ignored. + if (m.renamedFrom !== undefined) { + entry.renamedFrom = parseRef(m.renamedFrom, `models[${i}].renamedFrom`); + } + if (m.contextWindow !== undefined) { + if (typeof m.contextWindow !== "number" || !(m.contextWindow > 0)) { + throw badRequest(`models[${i}].contextWindow 必须是正数。`); + } + entry.contextWindow = m.contextWindow; + } + if (m.clientType !== undefined) { + if (typeof m.clientType !== "string" || m.clientType.length > 64) { + throw badRequest(`models[${i}].clientType 必须是长度不超过 64 的字符串。`); + } + // An empty string is treated as "unspecified", leaving AgentHub to infer it from modelId. + if (m.clientType) entry.clientType = m.clientType; + } + if (m.vision !== undefined) { + if (typeof m.vision !== "boolean") { + throw badRequest(`models[${i}].vision 必须是布尔值。`); + } + entry.vision = m.vision; + } + if (m.pricing !== undefined) { + const p = m.pricing as Record; + if (p === null || typeof p !== "object" || Array.isArray(p)) { + throw badRequest(`models[${i}].pricing 必须是对象。`); + } + for (const key of ["cacheRead", "cacheWrite", "output"] as const) { + const v = p[key]; + if (typeof v !== "number" || !Number.isFinite(v) || v < 0) { + throw badRequest(`models[${i}].pricing.${key} 必须是非负数字。`); + } + } + entry.pricing = { + cacheRead: p.cacheRead as number, + cacheWrite: p.cacheWrite as number, + output: p.output as number, + }; + } + if (m.apiKey !== undefined) { + if (typeof m.apiKey !== "string" || m.apiKey.length === 0) { + throw badRequest(`models[${i}].apiKey 必须是非空字符串。`); + } + entry.apiKey = m.apiKey; + } + if (m.clearApiKey !== undefined) { + if (typeof m.clearApiKey !== "boolean") { + throw badRequest(`models[${i}].clearApiKey 必须是布尔值。`); + } + entry.clearApiKey = m.clearApiKey; + } + if (m.baseUrl !== undefined) { + if (m.baseUrl !== null && typeof m.baseUrl !== "string") { + throw badRequest(`models[${i}].baseUrl 必须是字符串或 null。`); + } + entry.baseUrl = m.baseUrl as string | null; + } + return entry; + }); + const req: ModelsUpdateRequest = { models }; + if (body.defaultModel !== undefined) { + req.defaultModel = parseRef(body.defaultModel, "defaultModel"); + } + if (body.visionModel !== undefined) { + req.visionModel = parseRef(body.visionModel, "visionModel"); + } + return req; +} + +export function modelsRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + // Defensive id validation (FD-4). + const projectId = requireValidId(c, "projectId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + return c.json(await deps.projectConfigService.getModels(projectId)); + }); + + app.put("/", async (c) => { + const projectId = requireValidId(c, "projectId"); + deps.projectService.requireProjectOwner(c.var.user.userId, projectId); + const req = parseModelsUpdate(await readJson(c)); + return c.json(await deps.projectConfigService.updateModels(projectId, req)); + }); + + // Connectivity test (owner): the model reference `(provider, modelId)` is sent as a pair + // in the request body; sends one minimal request using that model's config. May include + // not-yet-saved apiKey / baseUrl / clientType — when the model isn't in the config yet + // (adding a custom model), everything is taken from the request body. + app.post("/test", async (c) => { + const projectId = requireValidId(c, "projectId"); + deps.projectService.requireProjectOwner(c.var.user.userId, projectId); + const body = await readJson(c); + const req: ModelTestRequest = { + provider: requireString(body, "provider", { minLen: 1, maxLen: 64 }), + modelId: requireString(body, "modelId", { minLen: 1, maxLen: 200 }), + }; + if (body.apiKey !== undefined) { + if (typeof body.apiKey !== "string") throw badRequest("apiKey 必须是字符串。"); + if (body.apiKey) req.apiKey = body.apiKey; + } + if (body.clearApiKey !== undefined) { + if (typeof body.clearApiKey !== "boolean") throw badRequest("clearApiKey 必须是布尔值。"); + req.clearApiKey = body.clearApiKey; + } + // null = explicit clear (test against the draft, don't fall back to the stored value); empty string is treated as null. + if (body.baseUrl !== undefined) { + if (body.baseUrl !== null && typeof body.baseUrl !== "string") { + throw badRequest("baseUrl 必须是字符串或 null。"); + } + req.baseUrl = body.baseUrl ? body.baseUrl : null; + } + if (body.clientType !== undefined) { + if (typeof body.clientType !== "string") throw badRequest("clientType 必须是字符串。"); + if (body.clientType) req.clientType = body.clientType; + } + return c.json(await deps.projectConfigService.testModel(projectId, req)); + }); + + return app; +} diff --git a/packages/server/src/http/routes/projects.ts b/packages/server/src/http/routes/projects.ts new file mode 100644 index 0000000..5536dc1 --- /dev/null +++ b/packages/server/src/http/routes/projects.ts @@ -0,0 +1,33 @@ +/** + * Project routes: GET|POST /api/projects, DELETE /api/projects/:p. + */ +import { Hono } from "hono"; +import type { ProjectCreateResponse, ProjectsResponse } from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { optionalString, readJson, requireString, requireValidId } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +export function projectsRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + const projects = await deps.projectService.listProjects(c.var.user.userId); + return c.json({ projects } satisfies ProjectsResponse); + }); + + app.post("/", async (c) => { + const body = await readJson(c); + const projectId = requireString(body, "projectId", { label: "projectId" }); + const name = optionalString(body, "name", { minLen: 1, maxLen: 100, label: "name" }); + const project = await deps.projectService.createProject(c.var.user, projectId, name); + return c.json({ project } satisfies ProjectCreateResponse, 201); + }); + + app.delete("/:projectId", async (c) => { + // Defensive id validation (FD-4): deleteProject constructs the project directory path and recursively deletes it. + await deps.projectService.deleteProject(c.var.user.userId, requireValidId(c, "projectId")); + return c.body(null, 204); + }); + + return app; +} diff --git a/packages/server/src/http/routes/schedules.ts b/packages/server/src/http/routes/schedules.ts new file mode 100644 index 0000000..8929308 --- /dev/null +++ b/packages/server/src/http/routes/schedules.ts @@ -0,0 +1,243 @@ +/** + * Schedule routes: + * GET|POST /api/projects/:p/agents/:a/schedules + * GET|PUT|DELETE /api/projects/:p/agents/:a/schedules/:name (name is the file name) + * Any member can read; only the owner can modify. The file is declarative intent: + * POST/PUT fully replace the file, validation always goes through parseScheduleFile + * (same rules as hand-edited files), and writes take effect immediately via reconciliation. + */ +import { createHash } from "node:crypto"; +import { Hono } from "hono"; +import { isValidId } from "@prismshadow/penguin-core"; +import type { ScheduleItem, ScheduleStatus, SchedulesResponse } from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import type { AppDeps } from "../../app.js"; +import { HttpError } from "../errors.js"; +import { + badRequest, + optionalString, + readJson, + requireString, + requireValidId, +} from "../validate.js"; +import type { ScheduleDefinition } from "../../runtime/schedule-file.js"; +import { + latestSlotAt, + nextSlotAfter, + parseScheduleFile, + slotInWindow, +} from "../../runtime/schedule-file.js"; +import type { ScheduleStateRow } from "../../db/repos/schedules.js"; +import { + deleteScheduleFile, + readScheduleFile, + serializeSchedule, + validateScheduleModelRef, + writeScheduleFile, +} from "../../runtime/schedule-store.js"; + +/** Validate and shape the POST/PUT request body into file fields (semantic validation is left to parseScheduleFile). */ +function parseUpsertBody(body: Record): { + prompt: string; + enabled: boolean; + startAt: string; + period?: string; + endAt?: string; + sessionId?: string; + workspace?: string; + modelId?: string; + provider?: string; +} { + if (typeof body.enabled !== "boolean") throw badRequest("enabled 必须是布尔值。"); + const prompt = requireString(body, "prompt", { minLen: 1, maxLen: 100_000 }); + const startAt = requireString(body, "startAt", { minLen: 1, maxLen: 100 }); + const period = optionalString(body, "period", { minLen: 1, maxLen: 20 }); + const endAt = optionalString(body, "endAt", { minLen: 1, maxLen: 100 }); + const sessionId = optionalString(body, "sessionId", { minLen: 1, maxLen: 200 }); + const workspace = optionalString(body, "workspace", { minLen: 1, maxLen: 4096 }); + const modelId = optionalString(body, "modelId", { minLen: 1, maxLen: 200 }); + const provider = optionalString(body, "provider", { minLen: 1, maxLen: 64 }); + return { + prompt, + enabled: body.enabled, + startAt, + ...(period !== undefined ? { period } : {}), + ...(endAt !== undefined ? { endAt } : {}), + ...(sessionId !== undefined ? { sessionId } : {}), + ...(workspace !== undefined ? { workspace } : {}), + ...(modelId !== undefined ? { modelId } : {}), + ...(provider !== undefined ? { provider } : {}), + }; +} + +/** Next scheduled fire time: none when disabled/invalid/done/missed; an undigested due slot counts as-is. */ +function nextFireAt( + def: ScheduleDefinition, + state: ScheduleStateRow, + nowMs: number, +): string | undefined { + if (!def.enabled || state.invalidReason !== null) return undefined; + if (def.periodMs === undefined && (state.firedOnce || state.missed)) return undefined; + const due = latestSlotAt(def, nowMs); + if ( + due !== null && + slotInWindow(def, due) && + (state.lastSlotMs === null || due > state.lastSlotMs) + ) { + return new Date(due).toISOString(); + } + const next = nextSlotAfter(def, nowMs); + return next !== null ? new Date(next).toISOString() : undefined; +} + +/** Displayed status precedence: invalid > done/missed (one-shot) > expired > enabled flag. */ +function statusOf(def: ScheduleDefinition, state: ScheduleStateRow, nowMs: number): ScheduleStatus { + if (state.invalidReason !== null) return "invalid"; + if (def.periodMs === undefined && state.firedOnce) return "done"; + if (def.periodMs === undefined && state.missed) return "missed"; + if (def.endAtMs !== undefined && nowMs > def.endAtMs) return "expired"; + return def.enabled ? "active" : "disabled"; +} + +function toItem( + def: ScheduleDefinition, + state: ScheduleStateRow, + queued: boolean, + nowMs: number, +): ScheduleItem { + const next = nextFireAt(def, state, nowMs); + return { + name: def.name, + prompt: def.prompt, + enabled: def.enabled, + startAt: def.startAt, + ...(def.period !== undefined ? { period: def.period } : {}), + ...(def.endAt !== undefined ? { endAt: def.endAt } : {}), + ...(def.sessionId !== undefined ? { sessionId: def.sessionId } : {}), + ...(def.workspace !== undefined ? { workspace: def.workspace } : {}), + ...(def.modelId !== undefined ? { modelId: def.modelId } : {}), + ...(def.provider !== undefined ? { provider: def.provider } : {}), + status: statusOf(def, state, nowMs), + ...(state.invalidReason !== null ? { invalidReason: state.invalidReason } : {}), + ...(next !== undefined ? { nextFireAt: next } : {}), + ...(state.lastFiredAt !== null ? { lastFiredAt: state.lastFiredAt } : {}), + queued, + ...(state.creatorUserId !== null ? { creatorUserId: state.creatorUserId } : {}), + }; +} + +/** Schedule name in the path: same character rules as directories/files, validated before any path construction. */ +function requireScheduleName(raw: string | undefined): string { + if (!raw || !isValidId(raw)) throw badRequest("定时任务名非法。"); + return raw; +} + +export function scheduleRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + const { entries, invalid } = await deps.scheduler.listAgent(projectId, agentId); + const nowMs = Date.now(); + const res: SchedulesResponse = { + schedules: entries.map((e) => toItem(e.def, e.state, e.queued, nowMs)), + invalidFiles: invalid, + }; + return c.json(res); + }); + + app.post("/", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectOwner(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + const body = await readJson(c); + const name = requireScheduleName(requireString(body, "name", { minLen: 1, maxLen: 100 })); + if (await readScheduleFile(deps.config.root, projectId, agentId, name)) { + throw new HttpError(409, "schedule_exists", `定时任务已存在:${name}`); + } + await upsert(deps, c.var.user.userId, projectId, agentId, name, body); + return c.json(await readItem(deps, projectId, agentId, name), 201); + }); + + app.get("/:name", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + const name = requireScheduleName(c.req.param("name")); + const item = await readItem(deps, projectId, agentId, name); + return c.json(item); + }); + + app.put("/:name", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectOwner(c.var.user.userId, projectId); + const name = requireScheduleName(c.req.param("name")); + if (!(await readScheduleFile(deps.config.root, projectId, agentId, name))) { + throw new HttpError(404, "schedule_not_found", `定时任务不存在:${name}`); + } + const body = await readJson(c); + await upsert(deps, c.var.user.userId, projectId, agentId, name, body); + return c.json(await readItem(deps, projectId, agentId, name)); + }); + + app.delete("/:name", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectOwner(c.var.user.userId, projectId); + const name = requireScheduleName(c.req.param("name")); + const removed = await deleteScheduleFile(deps.config.root, projectId, agentId, name); + if (!removed) throw new HttpError(404, "schedule_not_found", `定时任务不存在:${name}`); + deps.scheduler.dropEntry(projectId, agentId, name); + return c.body(null, 204); + }); + + return app; +} + +/** Write + register creator + reconcile immediately (API changes take effect right away). */ +async function upsert( + deps: AppDeps, + userId: string, + projectId: string, + agentId: string, + name: string, + body: Record, +): Promise { + const fields = parseUpsertBody(body); + const raw = serializeSchedule(fields); + const parsed = parseScheduleFile(name, raw); + if (!parsed.ok) throw badRequest(`定时任务配置非法:${parsed.error}`); + // At save time, verify the model reference resolves (resolveModelRef semantics; same rules as reconciliation) so we never persist a broken file. + const refError = await validateScheduleModelRef(deps.config.root, projectId, parsed.def); + if (refError !== null) throw badRequest(`定时任务配置非法:${refError}`); + await writeScheduleFile(deps.config.root, projectId, agentId, name, raw); + // Creator attribution: the API writer is the creator (falls back to the Project owner only for hand-edited files). + deps.schedulesRepo.registerOrSync({ + projectId, + agentId, + name, + startAtMs: parsed.def.startAtMs, + defHash: createHash("sha1").update(raw).digest("hex"), + creatorUserId: userId, + }); + await deps.scheduler.reconcileAgent(projectId, agentId); +} + +async function readItem( + deps: AppDeps, + projectId: string, + agentId: string, + name: string, +): Promise { + const { entries, invalid } = await deps.scheduler.listAgent(projectId, agentId); + const entry = entries.find((e) => e.def.name === name); + if (entry) return toItem(entry.def, entry.state, entry.queued, Date.now()); + const bad = invalid.find((i) => i.name === name); + if (bad) throw badRequest(`定时任务文件非法:${bad.error}`); + throw new HttpError(404, "schedule_not_found", `定时任务不存在:${name}`); +} diff --git a/packages/server/src/http/routes/sessions.ts b/packages/server/src/http/routes/sessions.ts new file mode 100644 index 0000000..5459846 --- /dev/null +++ b/packages/server/src/http/routes/sessions.ts @@ -0,0 +1,430 @@ +/** + * Session routes. + * + * Two entry groups: + * - Agent-level: GET|POST /api/projects/:p/agents/:a/sessions (list including run state / create); + * - Session-level: /api/sessions/:sessionId/* (no projectId; looks up project_id via the + * sessions index, then goes through requireProjectAccess; 404 if the index has no such Session). + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { Hono } from "hono"; +import type { Context } from "hono"; +import { imageUrlMessage, scratchpadDir, userText } from "@prismshadow/penguin-core"; +import type { OmniMessage } from "@prismshadow/penguin-core"; +import type { + ApprovalMode, + FilesStatResponse, + MessagesResponse, + ServerEvent, + SessionCreateResponse, + SessionResponse, + SessionsResponse, + TaskCreateResponse, +} from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import type { SessionRow } from "../../db/repos/sessions.js"; +import { assertWorkspaceAllowed } from "../../services/workspace-guard.js"; +import { HttpError } from "../errors.js"; +import { sseEndpoint } from "../sse.js"; +import { + badRequest, + optionalEnum, + optionalString, + paginationQuery, + pathParam, + positiveIntParam, + readJson, + requireEnum, + requireValidId, +} from "../validate.js"; +import type { AppDeps } from "../../app.js"; +import { MAX_UPLOAD_BYTES } from "../../services/workspace-files-service.js"; + +/** Max title length for manual renames: looser than the auto-generated 30-char limit, to accommodate users' own organizing conventions. */ +const SESSION_TITLE_MAX = 120; + +/** Max path count and per-path length for a single files/stat check (message file-card candidates never exceed this scale). */ +const STAT_MAX_PATHS = 100; +const STAT_MAX_PATH_LEN = 512; + +const APPROVAL_MODES: readonly ApprovalMode[] = [ + "allow-all", + "deny-all", + "read-only", + "always-ask", +]; + +/** Validate Prompt input parts: text or image (data: / http(s) URL). */ +function parseTaskInput(body: Record): OmniMessage[] { + const input = body.input; + if (!Array.isArray(input) || input.length === 0) { + throw badRequest("input 必须是至少包含一项的数组。"); + } + return input.map((item, i) => { + if (item === null || typeof item !== "object" || Array.isArray(item)) { + throw badRequest(`input[${i}] 必须是对象。`); + } + const part = item as Record; + if (part.type === "text") { + if (typeof part.text !== "string" || part.text.length === 0) { + throw badRequest(`input[${i}].text 必须是非空字符串。`); + } + return userText(part.text); + } + if (part.type === "image_url") { + const url = part.imageUrl; + if ( + typeof url !== "string" || + !(url.startsWith("data:") || url.startsWith("http://") || url.startsWith("https://")) + ) { + throw badRequest(`input[${i}].imageUrl 仅支持 data: 或 http(s) URL。`); + } + return imageUrlMessage(url); + } + throw badRequest(`input[${i}].type 必须是 text / image_url 之一。`); + }); +} + +/** Agent-level entry: /api/projects/:p/agents/:a/sessions. */ +export function agentSessionsRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + // Id validity is checked before any path is constructed (FD-4: guards against agentId path traversal across Projects). + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + const sessions = await deps.sessionService.listSessions(projectId, agentId); + return c.json({ sessions } satisfies SessionsResponse); + }); + + app.post("/", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + const body = await readJson(c); + const modelId = optionalString(body, "modelId", { minLen: 1, label: "modelId" }); + const provider = optionalString(body, "provider", { minLen: 1, label: "provider" }); + // Model reference is submitted as a pair: provider can't appear without modelId (core does the same validation; this catches it early). + if (provider !== undefined && modelId === undefined) { + throw badRequest("指定了 provider 却未指定 modelId:模型引用须成对给出。"); + } + const approvalMode = optionalEnum(body, "approvalMode", APPROVAL_MODES); + let workspace = optionalString(body, "workspace", { minLen: 1, label: "workspace" }); + if (workspace !== undefined) { + // An explicitly specified Workspace must be an existing directory (never auto-created); reachability is determined by file permissions. + workspace = await assertWorkspaceAllowed({ workspace }); + } + const session = await deps.sessionService.createSession({ + projectId, + agentId, + ...(modelId !== undefined ? { modelId } : {}), + ...(provider !== undefined ? { provider } : {}), + ...(workspace !== undefined ? { workspace } : {}), + ...(approvalMode !== undefined ? { approvalMode } : {}), + }); + return c.json({ session } satisfies SessionCreateResponse, 201); + }); + + return app; +} + +/** Session-level entry point: /api/sessions/:sessionId/*. */ +export function sessionsRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + /** Look up ownership and check access (404 if the index has no such Session, or access is denied — never leaking existence). */ + const resolveSession = (c: Context): SessionRow => { + const sessionId = c.req.param("sessionId"); + const row = sessionId ? deps.sessionsRepo.findById(sessionId) : null; + if (!row) { + throw new HttpError(404, "session_not_found", "Session 不存在或无权访问。"); + } + try { + deps.projectService.requireProjectAccess(c.var.user.userId, row.projectId); + } catch { + throw new HttpError(404, "session_not_found", "Session 不存在或无权访问。"); + } + return row; + }; + + app.get("/:sessionId", async (c) => { + const row = resolveSession(c); + const hasTrace = await deps.sessionService.hasTrace(row); + return c.json({ session: deps.sessionService.toInfo(row, hasTrace) } satisfies SessionResponse); + }); + + app.patch("/:sessionId", async (c) => { + const row = resolveSession(c); + const body = await readJson(c); + const approvalMode = optionalEnum(body, "approvalMode", APPROVAL_MODES); + const archivedRaw = (body as Record).archived; + const archived = typeof archivedRaw === "boolean" ? archivedRaw : undefined; + const titleRaw = (body as Record).title; + let title: string | undefined; + if (titleRaw !== undefined) { + if (typeof titleRaw !== "string") { + throw new HttpError(400, "invalid_title", "title 必须是字符串。"); + } + title = titleRaw.trim(); + if (!title || title.length > SESSION_TITLE_MAX) { + throw new HttpError(400, "invalid_title", `title 需为 1–${SESSION_TITLE_MAX} 个字符。`); + } + } + if (approvalMode === undefined && archived === undefined && title === undefined) { + throw new HttpError(400, "no_update", "缺少可更新字段(approvalMode / archived / title)。"); + } + let updated: SessionRow = { ...row }; + if (title !== undefined) { + // Manual renaming takes priority over auto-generation: TitleGenerator only persists a title while it's still NULL. + deps.sessionsRepo.updateTitle(row.sessionId, title); + updated = { ...updated, title }; + } + if (approvalMode !== undefined) { + // Takes effect immediately: a running approve callback re-reads the DB on every decision. + deps.sessionsRepo.updateApprovalMode(row.sessionId, approvalMode); + updated = { ...updated, approvalMode }; + } + if (archived !== undefined) { + const at = archived ? new Date().toISOString() : null; + deps.sessionsRepo.setArchived(row.sessionId, at); + updated = { ...updated, archivedAt: at }; + } + const hasTrace = await deps.sessionService.hasTrace(updated); + return c.json({ + session: deps.sessionService.toInfo(updated, hasTrace), + } satisfies SessionResponse); + }); + + app.delete("/:sessionId", async (c) => { + const row = resolveSession(c); + // Mark as being deleted and converge active runs (beginSessionDeletion): new + // Tasks/compactions are always rejected with 409 during this window + // (assertSessionNotDeleting), preventing the race where a new task recreates the + // entry and Trace after abort but before the files are deleted, reviving an + // already-deleted Session. Interrupt cleanup writes the Trace asynchronously, so we + // wait for it to finish (≤5s cap) before deleting the files and index row; the + // being-deleted marker is cleared once deletion finishes (success or failure). + const runnings = deps.manager.beginSessionDeletion(row.sessionId); + try { + if (runnings.length > 0) { + await Promise.race([ + Promise.allSettled(runnings).then(() => undefined), + new Promise((resolve) => setTimeout(resolve, 5000).unref?.()), + ]); + } + await deps.traceService.deleteSessionTraces(row.projectId, row.agentId, row.sessionId); + // The session-level scratchpad (model temp files + input images saved to disk for image-unsupported models) is deleted along with the session. + await fs.rm( + path.join(scratchpadDir(deps.config.root, row.projectId, row.agentId), row.sessionId), + { recursive: true, force: true }, + ); + deps.sessionsRepo.deleteById(row.sessionId); + } finally { + deps.manager.endSessionDeletion(row.sessionId); + } + return c.body(null, 204); + }); + + // Session scratchpad files (e.g. input images saved to disk for image-unsupported + // models): read by filename, so the conversation UI can render a message's + // "[attached image: ]" attachment line back into an image. Restricted to this + // session's own scratchpad directory (the filename must not contain a path + // separator, blocking traversal); filenames include a timestamp and are globally + // unique, so the response is marked immutable and long-cacheable. + app.get("/:sessionId/scratchpad/:fileName", async (c) => { + const row = resolveSession(c); + const fileName = c.req.param("fileName") ?? ""; + if (!/^[A-Za-z0-9._-]+$/.test(fileName) || fileName.includes("..")) { + throw new HttpError(404, "file_not_found", "文件不存在。"); + } + const filePath = path.join( + scratchpadDir(deps.config.root, row.projectId, row.agentId), + row.sessionId, + fileName, + ); + let bytes: Buffer; + try { + bytes = await fs.readFile(filePath); + } catch { + throw new HttpError(404, "file_not_found", "文件不存在。"); + } + const MIME_BY_EXT: Record = { + ".png": "image/png", + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".gif": "image/gif", + ".webp": "image/webp", + }; + const mime = MIME_BY_EXT[path.extname(fileName).toLowerCase()] ?? "application/octet-stream"; + return c.body(new Uint8Array(bytes), 200, { + "content-type": mime, + "cache-control": "private, max-age=31536000, immutable", + }); + }); + + app.get("/:sessionId/messages", async (c) => { + const row = resolveSession(c); + const messages = await deps.traceService.readMessages( + row.projectId, + row.agentId, + row.sessionId, + ); + return c.json({ messages } satisfies MessagesResponse); + }); + + app.get("/:sessionId/stream", (c) => { + const row = resolveSession(c); + const channel = deps.channels.get(row.sessionId); + // FD-1: the first event of every new subscription (including reconnects and resync + // rebuilds) is always a snapshot of the current running state — the frontend treats + // this as authoritative, eliminating input-area lockup or premature Task closure + // caused by a stale running/idle in the list; followed by replaying all still-pending + // approval requests. + const initialEvents: ServerEvent[] = [ + { type: "task_state", state: deps.manager.statusOf(row.sessionId) }, + ...deps.manager.pendingApprovals(row.sessionId).map((p) => ({ + type: "approval_request" as const, + toolCall: p.toolCall, + ...(p.origin !== undefined ? { origin: p.origin } : {}), + })), + ]; + return sseEndpoint(c, channel, { initialEvents }); + }); + + app.post("/:sessionId/tasks", async (c) => { + const row = resolveSession(c); + const input = parseTaskInput(await readJson(c)); + // 202: the Task executes on the server, decoupled from the SSE connection; sessionId is the current actual id (the new id after self-heal). + const { sessionId } = await deps.manager.startTask(row.sessionId, input); + return c.json({ sessionId } satisfies TaskCreateResponse, 202); + }); + + app.post("/:sessionId/approvals/:toolCallId", async (c) => { + const row = resolveSession(c); + const body = await readJson(c); + const decision = requireEnum(body, "decision", ["allow", "deny"] as const); + const ok = deps.manager.decideApproval(row.sessionId, pathParam(c, "toolCallId"), decision); + if (!ok) { + throw new HttpError(404, "approval_not_found", "该审批不存在或已被决定。"); + } + return c.body(null, 204); + }); + + app.post("/:sessionId/abort", (c) => { + const row = resolveSession(c); + const aborted = deps.manager.abortTask(row.sessionId); + // No Task in progress → 204 no-op; interrupt was triggered → 202 (wrap-up is completed by the SDK's "interrupt cleanup"). + return c.body(null, aborted ? 202 : 204); + }); + + app.post("/:sessionId/compact", async (c) => { + const row = resolveSession(c); + const { sessionId } = await deps.manager.startCompact(row.sessionId); + return c.json({ sessionId } satisfies TaskCreateResponse, 202); + }); + + // —— Workspace file browsing (Files tab) —— + + app.get("/:sessionId/files", async (c) => { + const row = resolveSession(c); + const rel = c.req.query("path") ?? ""; + return c.json(await deps.workspaceFiles.list(row.workspace, rel)); + }); + + app.get("/:sessionId/files/content", async (c) => { + const row = resolveSession(c); + const rel = c.req.query("path") ?? ""; + const download = c.req.query("download") === "1"; + const { data, fileName, contentType, scriptable } = await deps.workspaceFiles.read( + row.workspace, + rel, + ); + const disposition = download ? "attachment" : "inline"; + // Same-origin XSS defense: html/svg inline previews are always returned as plain + // text (Workspace files may be Agent-generated and untrusted); downloads + // (attachment) keep the real content type. Paired with nosniff to prevent MIME + // sniffing from undoing this. + const effectiveType = !download && scriptable ? "text/plain; charset=utf-8" : contentType; + return new Response(new Uint8Array(data), { + status: 200, + headers: { + "Content-Type": effectiveType, + "Content-Disposition": `${disposition}; filename*=UTF-8''${encodeURIComponent(fileName)}`, + "X-Content-Type-Options": "nosniff", + }, + }); + }); + + // Bulk existence check (message file cards list only files that actually exist): + // path-confinement resolution shares the same logic as files/content + // (WorkspaceFilesService.statExisting reuses resolveRead); out-of-bounds or + // resolution failures count as not-existing, always 200 — existence itself is the + // question being answered, and a 4xx would only leak confinement details. + app.post("/:sessionId/files/stat", async (c) => { + const row = resolveSession(c); + const body = await readJson(c); + const paths = body.paths; + if ( + !Array.isArray(paths) || + paths.length > STAT_MAX_PATHS || + !paths.every((p) => typeof p === "string" && p.length <= STAT_MAX_PATH_LEN) + ) { + throw badRequest( + `paths 必须是字符串数组(≤${STAT_MAX_PATHS} 项,每项 ≤${STAT_MAX_PATH_LEN} 字符)。`, + ); + } + const existing = await deps.workspaceFiles.statExisting(row.workspace, paths as string[]); + return c.json({ existing } satisfies FilesStatResponse); + }); + + app.put("/:sessionId/files/content", async (c) => { + const row = resolveSession(c); + const rel = c.req.query("path") ?? ""; + const body = await readJson(c); + if (typeof body.dataBase64 !== "string") { + throw badRequest("dataBase64 必须是 base64 字符串。"); + } + const data = Buffer.from(body.dataBase64, "base64"); + if (data.length > MAX_UPLOAD_BYTES) { + throw new HttpError(413, "file_too_large", "上传文件超过 14MB 上限。"); + } + await deps.workspaceFiles.write(row.workspace, rel, data); + return c.body(null, 204); + }); + + app.get("/:sessionId/traces", async (c) => { + const row = resolveSession(c); + const files = await deps.traceService.listTraceFiles(row.projectId, row.agentId, row.sessionId); + return c.json({ files }); + }); + + app.get("/:sessionId/traces/:index", async (c) => { + const row = resolveSession(c); + const index = positiveIntParam(c, "index"); + const { offset, limit } = paginationQuery(c); + return c.json( + await deps.traceService.readEvents( + row.projectId, + row.agentId, + row.sessionId, + index, + offset, + limit, + ), + ); + }); + + app.get("/:sessionId/traces/:index/analysis", async (c) => { + const row = resolveSession(c); + const index = positiveIntParam(c, "index"); + return c.json( + await deps.traceService.analyze(row.projectId, row.agentId, row.sessionId, index), + ); + }); + + return app; +} diff --git a/packages/server/src/http/routes/skills.ts b/packages/server/src/http/routes/skills.ts new file mode 100644 index 0000000..fe4debd --- /dev/null +++ b/packages/server/src/http/routes/skills.ts @@ -0,0 +1,139 @@ +/** + * Skill library & Agent-installed-Skills routes: + * GET /api/skills # library groups & metadata (any logged-in user) + * GET|POST /api/projects/:p/agents/:a/skills # installed list / install from library (any member) + * DELETE /api/projects/:p/agents/:a/skills/:name # uninstall (any member) + * Installing writes the library's SKILL.md verbatim to agent_state/skills//; + * reinstalling overwrites with the library content (i.e. an update). The scope is small + * enough to skip a service layer — routes call core's disk-writing functions directly. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { Hono } from "hono"; +import { + installSkill, + listInstalledSkills, + removeSkill, + skillsDir, +} from "@prismshadow/penguin-core"; +import { librarySkill, loadSkillGroups } from "@prismshadow/penguin-skills"; +import type { LibrarySkill, SkillMetadata } from "@prismshadow/penguin-skills"; +import type { + AgentSkillsResponse, + SkillLibraryResponse, + SkillMetadataItem, +} from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import type { AppDeps } from "../../app.js"; +import { HttpError } from "../errors.js"; +import { badRequest, readJson, requireValidId } from "../validate.js"; + +/** + * Strips the content off a LibrarySkill: the API only sends metadata; the full body is + * written to disk on install and read by the model on demand. The optional short + * description (shortDescription(Zh)) and custom icon (icon.svg source) are conditionally + * passed through — both the library side (LibrarySkill) and the installed side (core + * InstalledSkill) carry these fields. + */ +function toMetadataItem(skill: SkillMetadata & { icon?: string }): SkillMetadataItem { + return { + name: skill.name, + description: skill.description, + ...(skill.shortDescription !== undefined ? { shortDescription: skill.shortDescription } : {}), + ...(skill.shortDescriptionZh !== undefined + ? { shortDescriptionZh: skill.shortDescriptionZh } + : {}), + ...(skill.icon !== undefined ? { icon: skill.icon } : {}), + version: skill.version, + updated: skill.updated, + }; +} + +/** Library listing response: the files are the source of truth — read and parse the library directory fresh on every request (files are small, requests infrequent, no caching needed). */ +function libraryResponse(): SkillLibraryResponse { + return { + groups: loadSkillGroups().map((group) => ({ + id: group.id, + title: group.title, + ...(group.titleZh !== undefined ? { titleZh: group.titleZh } : {}), + skills: group.skills.map(toMetadataItem), + })), + }; +} + +/** Validate the POST request body: names must be a non-empty array of strings. */ +function parseInstallNames(body: Record): string[] { + if (!Array.isArray(body.names) || body.names.length === 0) { + throw badRequest("names 必须是非空数组。"); + } + return body.names.map((v, i) => { + if (typeof v !== "string" || v.length === 0) { + throw badRequest(`names[${i}] 必须是非空字符串。`); + } + return v; + }); +} + +/** GET /api/skills: Skill library groups & metadata (any logged-in user; no Project check). */ +export function skillLibraryRoutes(): Hono { + const app = new Hono(); + app.get("/", (c) => c.json(libraryResponse())); + return app; +} + +/** /api/projects/:p/agents/:a/skills: read, install, and uninstall are all Project-member operations. */ +export function agentSkillsRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + const listResponse = async ( + projectId: string, + agentId: string, + ): Promise => ({ + skills: (await listInstalledSkills(deps.config.root, projectId, agentId)).map(toMetadataItem), + }); + + app.get("/", async (c) => { + // Defensive id validation happens before any path construction (FD-4: prevents path traversal for cross-Project privilege escalation). + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + return c.json(await listResponse(projectId, agentId)); + }); + + app.post("/", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + const names = parseInstallNames(await readJson(c)); + // Verify all names up front before writing anything: if any name isn't in the library, reject the whole request rather than leaving a half-installed state. + const skills: LibrarySkill[] = names.map((name) => { + const skill = librarySkill(name); + if (!skill) throw new HttpError(404, "unknown_skill", `Skill 库中不存在:${name}`); + return skill; + }); + for (const skill of skills) { + await installSkill(deps.config.root, projectId, agentId, skill); + } + return c.json(await listResponse(projectId, agentId), 201); + }); + + app.delete("/:name", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + const name = requireValidId(c, "name"); + // Installed-check uses the same criterion as listInstalledSkills: skills//SKILL.md exists. + const file = path.join(skillsDir(deps.config.root, projectId, agentId), name, "SKILL.md"); + try { + await fs.access(file); + } catch { + throw new HttpError(404, "not_found", `Skill 未安装:${name}`); + } + await removeSkill(deps.config.root, projectId, agentId, name); + return c.body(null, 204); + }); + + return app; +} diff --git a/packages/server/src/http/routes/usage.ts b/packages/server/src/http/routes/usage.ts new file mode 100644 index 0000000..f98fdde --- /dev/null +++ b/packages/server/src/http/routes/usage.ts @@ -0,0 +1,48 @@ +/** + * Usage statistics routes: + * GET /api/projects/:p/usage?from&to&groupBy&agentId&provider&modelId + * (model filter is paired: provider and modelId are given together). + */ +import { Hono } from "hono"; +import type { UsageGroupBy } from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { badRequest, optionalDateParam, requireValidId } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +const GROUP_BYS: readonly UsageGroupBy[] = ["date", "agent", "model", "session"]; + +export function usageRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + // Defensive id validation (FD-4). + const projectId = requireValidId(c, "projectId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + const groupByRaw = c.req.query("groupBy") ?? "date"; + if (!(GROUP_BYS as readonly string[]).includes(groupByRaw)) { + throw badRequest(`groupBy 必须是 ${GROUP_BYS.join(" / ")} 之一。`); + } + const from = optionalDateParam(c.req.query("from"), "from"); + const to = optionalDateParam(c.req.query("to"), "to"); + const agentId = c.req.query("agentId"); + const provider = c.req.query("provider"); + const modelId = c.req.query("modelId"); + return c.json( + await deps.usageService.query(projectId, { + groupBy: groupByRaw as UsageGroupBy, + // Unattributed errors (login failures, process crashes, etc. with no Project + // context) are visible only to admins: requireProjectAccess only guarantees + // "is a member of this Project" — a regular member seeing another tenant's errors + // would be a cross-tenant information leak. + includeGlobalErrors: c.var.user.isAdmin, + ...(from !== undefined ? { from } : {}), + ...(to !== undefined ? { to } : {}), + ...(agentId !== undefined && agentId !== "" ? { agentId } : {}), + ...(provider !== undefined && provider !== "" ? { provider } : {}), + ...(modelId !== undefined && modelId !== "" ? { modelId } : {}), + }), + ); + }); + + return app; +} diff --git a/packages/server/src/http/routes/vault.ts b/packages/server/src/http/routes/vault.ts new file mode 100644 index 0000000..90495ed --- /dev/null +++ b/packages/server/src/http/routes/vault.ts @@ -0,0 +1,58 @@ +/** + * Vault environment variable routes: + * GET|PUT /api/projects/:p/agents/:a/vault (Agent-level, agent_state/.vault.toml). + * Any member can read (values masked); only the owner can modify; 404 if the Agent doesn't exist. + */ +import { Hono } from "hono"; +import type { VaultEntryUpdate, VaultUpdateRequest } from "../../api/types.js"; +import type { AppEnv } from "../../auth/middleware.js"; +import { badRequest, readJson, requireString, requireValidId } from "../validate.js"; +import type { AppDeps } from "../../app.js"; + +/** Validate the PUT request body and shape it into a VaultUpdateRequest (semantic checks like key-name rules live in the service layer). */ +function parseVaultUpdate(body: Record): VaultUpdateRequest { + if (!Array.isArray(body.entries)) throw badRequest("entries 必须是数组。"); + const entries: VaultEntryUpdate[] = body.entries.map((item, i) => { + if (item === null || typeof item !== "object" || Array.isArray(item)) { + throw badRequest(`entries[${i}] 必须是对象。`); + } + const e = item as Record; + const entry: VaultEntryUpdate = { + key: requireString(e, "key", { minLen: 1, maxLen: 200, label: `entries[${i}].key` }), + }; + if (e.value !== undefined) { + if (typeof e.value !== "string" || e.value.length === 0) { + throw badRequest(`entries[${i}].value 必须是非空字符串。`); + } + entry.value = e.value; + } + return entry; + }); + return { entries }; +} + +export function vaultRoutes(deps: AppDeps): Hono { + const app = new Hono(); + + app.get("/", async (c) => { + // Defensive id validation happens before any path construction (FD-4: prevents agentId path traversal for cross-Project privilege escalation). + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + return c.json(await deps.agentConfigService.getVault(projectId, agentId)); + }); + + app.put("/", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + deps.projectService.requireProjectOwner(c.var.user.userId, projectId); + const req = parseVaultUpdate(await readJson(c)); + const res = await deps.agentConfigService.updateVault(projectId, agentId, req); + // Effective-value semantics: no hot update — an already-built runtime is neither + // evicted nor reloaded; the new value only applies to Sessions created or resumed + // afterward. + return c.json(res); + }); + + return app; +} diff --git a/packages/server/src/http/settle.ts b/packages/server/src/http/settle.ts new file mode 100644 index 0000000..91f770f --- /dev/null +++ b/packages/server/src/http/settle.ts @@ -0,0 +1,13 @@ +/** + * Convergent wait for delete routes (currently used by agents DELETE; sessions DELETE + * uses the same inline pattern and could later be unified onto this): waits for aborted + * runs to wind down, up to ms milliseconds; returns whether all of them settled within + * the window. The timer is unref'd so it never blocks process exit. + */ +export async function settleWithin(promises: Promise[], ms: number): Promise { + if (promises.length === 0) return true; + return Promise.race([ + Promise.allSettled(promises).then(() => true), + new Promise((resolve) => setTimeout(() => resolve(false), ms).unref?.()), + ]); +} diff --git a/packages/server/src/http/sse.ts b/packages/server/src/http/sse.ts new file mode 100644 index 0000000..99f3baa --- /dev/null +++ b/packages/server/src/http/sse.ts @@ -0,0 +1,94 @@ +/** + * SSE endpoint adapter: + * writes the runtime/channel event stream as an SSE response, shared by both the Session + * channel and the user channel. + * + * - Response headers: text/event-stream, no-cache, `X-Accel-Buffering: no` (disables + * buffering on reverse proxies); + * - Heartbeat: writes a `: ping` comment line every 20s; the connection is torn down on write failure; + * - Replay protocol: a fresh subscription without Last-Event-ID does not replay the buffer + * (history is served by the messages endpoint) — it only sends the initial events the + * caller supplied (pending approvals / hello). With a Last-Event-ID that hits the buffer, + * replay resumes from there; on a miss, `resync_required` is sent first, then the + * connection continues. + * Docs: /docs/server-api § "Delivery Guarantees". + */ +import type { Context } from "hono"; +import { streamSSE } from "hono/streaming"; +import type { ServerEvent } from "../api/types.js"; +import type { Channel, ChannelEvent, ChannelListener } from "../runtime/channel.js"; + +const HEARTBEAT_MS = 20_000; + +export interface SseEndpointOptions { + /** Initial server events to send privately after the subscription is established (in order): pending-approval replay / user channel hello. */ + initialEvents?: ServerEvent[]; +} + +/** Stream out a Channel as an SSE response. */ +export function sseEndpoint(c: Context, channel: Channel, opts: SseEndpointOptions = {}): Response { + const lastEventIdHeader = c.req.header("Last-Event-ID"); + c.header("X-Accel-Buffering", "no"); + c.header("Cache-Control", "no-cache"); + + return streamSSE(c, async (stream) => { + let closed = false; + let finish: () => void = () => {}; + const done = new Promise((resolve) => { + finish = () => { + if (closed) return; + closed = true; + resolve(); + }; + }); + + // Write serialization: SSE events must be written fully and in order; any write failure tears down the connection. + let chain: Promise = Promise.resolve(); + const enqueue = (write: () => Promise): void => { + chain = chain + .then(async () => { + if (!closed) await write(); + }) + .catch(() => finish()); + }; + const listener: ChannelListener = (evt: ChannelEvent) => { + enqueue(() => + stream.writeSSE({ + data: evt.data, + // Event id is an opaque string generated by the channel (`-`), passed through as-is. + id: evt.id, + ...(evt.event !== undefined ? { event: evt.event } : {}), + }), + ); + }; + + const unsubscribe = channel.subscribe(listener); + // Subscribe first (to avoid dropping events in a race with broadcasts), then replay + // synchronously — event order: buffered replay (or resync_required) -> initial events + // (task_state snapshot / pending approvals / hello) -> live stream. + if (lastEventIdHeader !== undefined) { + const replay = channel.replayAfter(lastEventIdHeader); + if (!replay.hit) { + const resync: ServerEvent = { type: "resync_required" }; + channel.sendTo(listener, resync, "server_event"); + } else { + for (const evt of replay.events) listener(evt); + } + } + for (const event of opts.initialEvents ?? []) { + channel.sendTo(listener, event, "server_event"); + } + + const heartbeat = setInterval(() => { + enqueue(() => stream.write(": ping\n\n")); + }, HEARTBEAT_MS); + + stream.onAbort(() => finish()); + try { + await done; + } finally { + clearInterval(heartbeat); + unsubscribe(); + } + }); +} diff --git a/packages/server/src/http/validate.ts b/packages/server/src/http/validate.ts new file mode 100644 index 0000000..3125611 --- /dev/null +++ b/packages/server/src/http/validate.ts @@ -0,0 +1,169 @@ +/** + * Hand-rolled request body validation helpers (fields follow TypeScript types; + * this adds a runtime safety net). + * + * No validation library: each helper checks one basic shape, throwing a 400 HttpError on failure. + */ +import type { Context } from "hono"; +import { isValidId } from "@prismshadow/penguin-core"; +import { HttpError } from "./errors.js"; + +export function badRequest(message: string): HttpError { + return new HttpError(400, "bad_request", message); +} + +/** + * Get a path parameter (under sub-route mounting, hono infers string | undefined; the + * route guarantees presence at runtime — treat a defensive missing value as 404). + */ +export function pathParam(c: Context, name: string): string { + const v = c.req.param(name); + if (v === undefined || v === "") { + throw new HttpError(404, "not_found", "路径参数缺失。"); + } + return v; +} + +/** + * Get a path parameter and validate the id (alphanumeric, underscore, and hyphen + * only, to prevent path traversal). Hono decodes URL-encoded `%2F` into a single path + * parameter; an id containing `/` or `..` passed straight into path construction could + * escape the resource directory (cross-Project privilege escalation). So validate right + * after reading the value — any invalid id is rejected with 404 (not leaking resource + * existence), before any service-layer or path-construction code runs. + */ +export function requireValidId(c: Context, name: string): string { + const v = pathParam(c, name); + if (!isValidId(v)) { + throw new HttpError(404, "not_found", "资源不存在或无权访问。"); + } + return v; +} + +/** Parse a positive-integer path parameter (e.g. Trace file index). */ +export function positiveIntParam(c: Context, name: string): number { + const v = Number.parseInt(pathParam(c, name), 10); + if (!Number.isInteger(v) || v < 1) throw badRequest(`${name} 必须是正整数。`); + return v; +} + +/** Parse Trace pagination query params: offset >= 0 (default 0), limit 1-1000 (default 200). */ +export function paginationQuery(c: Context): { offset: number; limit: number } { + const offset = Number.parseInt(c.req.query("offset") ?? "0", 10); + const limit = Number.parseInt(c.req.query("limit") ?? "200", 10); + if (!Number.isInteger(offset) || offset < 0) throw badRequest("offset 必须是非负整数。"); + if (!Number.isInteger(limit) || limit < 1 || limit > 1000) { + throw badRequest("limit 必须是 1~1000 的整数。"); + } + return { offset, limit }; +} + +/** Read the JSON request body (parse failure / non-object -> 400). */ +export async function readJson(c: Context): Promise> { + let body: unknown; + try { + body = await c.req.json(); + } catch { + throw badRequest("请求体必须是合法 JSON。"); + } + if (body === null || typeof body !== "object" || Array.isArray(body)) { + throw badRequest("请求体必须是 JSON 对象。"); + } + return body as Record; +} + +export interface StringRule { + minLen?: number; + maxLen?: number; + pattern?: RegExp; + /** Display name for the field in error messages (defaults to key). */ + label?: string; +} + +export function requireString( + obj: Record, + key: string, + rule: StringRule = {}, +): string { + const v = obj[key]; + const label = rule.label ?? key; + if (typeof v !== "string") throw badRequest(`${label} 必须是字符串。`); + if (rule.minLen !== undefined && v.length < rule.minLen) { + throw badRequest(`${label} 长度至少 ${rule.minLen} 个字符。`); + } + if (rule.maxLen !== undefined && v.length > rule.maxLen) { + throw badRequest(`${label} 长度不能超过 ${rule.maxLen} 个字符。`); + } + if (rule.pattern !== undefined && !rule.pattern.test(v)) { + throw badRequest(`${label} 格式不合法。`); + } + return v; +} + +export function optionalString( + obj: Record, + key: string, + rule: StringRule = {}, +): string | undefined { + if (obj[key] === undefined) return undefined; + return requireString(obj, key, rule); +} + +export function requireEnum( + obj: Record, + key: string, + values: readonly T[], + label = key, +): T { + const v = obj[key]; + if (typeof v !== "string" || !(values as readonly string[]).includes(v)) { + throw badRequest(`${label} 必须是 ${values.join(" / ")} 之一。`); + } + return v as T; +} + +export function optionalEnum( + obj: Record, + key: string, + values: readonly T[], + label = key, +): T | undefined { + if (obj[key] === undefined) return undefined; + return requireEnum(obj, key, values, label); +} + +export interface NumberRule { + /** Require positive or -1 (Agent runtime parameter convention: >0 active, -1 disabled). */ + positiveOrMinusOne?: boolean; + /** Require non-negative. */ + nonNegative?: boolean; + /** Require integer. */ + integer?: boolean; + label?: string; +} + +export function optionalNumber( + obj: Record, + key: string, + rule: NumberRule = {}, +): number | undefined { + const v = obj[key]; + if (v === undefined) return undefined; + const label = rule.label ?? key; + if (typeof v !== "number" || !Number.isFinite(v)) throw badRequest(`${label} 必须是数字。`); + if (rule.integer && !Number.isInteger(v)) throw badRequest(`${label} 必须是整数。`); + if (rule.positiveOrMinusOne && !(v > 0 || v === -1)) { + throw badRequest(`${label} 必须大于 0 或为 -1。`); + } + if (rule.nonNegative && v < 0) throw badRequest(`${label} 不能为负数。`); + return v; +} + +/** Validate a yyyy-mm-dd query parameter (defaults to undefined). */ +export function optionalDateParam(value: string | undefined, label: string): string | undefined { + if (value === undefined || value === "") return undefined; + if (!/^\d{4}-\d{2}-\d{2}$/.test(value)) { + throw badRequest(`${label} 必须是 YYYY-MM-DD 格式。`); + } + return value; +} diff --git a/packages/server/src/index.ts b/packages/server/src/index.ts new file mode 100644 index 0000000..c1f0da2 --- /dev/null +++ b/packages/server/src/index.ts @@ -0,0 +1,77 @@ +/** + * Server startup entry point: dotenv → config → assembly → listen + * → graceful shutdown. + * + * SIGINT / SIGTERM: interrupt all active runs (pending approvals converge to deny), wait + * ≤5s for wrap-up, then close HTTP and SQLite. Tests never go through this file + * (injected via app.request() instead). + * There's also a process-level error fallback (uncaughtException / unhandledRejection): + * persist + log, with the fatal one still shutting down per existing semantics (see the + * comment below). + */ +import { config as loadDotenv } from "dotenv"; +import { serve } from "@hono/node-server"; +import { buildAppDeps, createApp } from "./app.js"; +import { resolveServerConfig } from "./config.js"; + +loadDotenv({ quiet: true }); + +const config = resolveServerConfig(); +const deps = buildAppDeps(config); +const app = createApp(deps); + +// Built-in admin seed (idempotent): creates admin (initial password admin123) and adopts default_project when the users table is empty. +await deps.authService.seedAdmin(); + +// Schedule scheduler: startup reconciliation (missed, don't backfill) + periodic scan; only active while the server is running. +await deps.scheduler.start(); + +const server = serve({ fetch: app.fetch, hostname: config.host, port: config.port }, (info) => { + console.log(`penguin-server 已启动: http://${config.host}:${info.port}`); + console.log(`数据根目录: ${config.root}`); + console.log(`SQLite: ${config.dbPath}`); +}); + +let shuttingDown = false; +async function shutdown(signal: string, exitCode = 0): Promise { + if (shuttingDown) return; + shuttingDown = true; + console.log(`收到 ${signal},正在关停…`); + deps.scheduler.stop(); + await deps.manager.shutdown(5000); + deps.channels.dispose(); + server.close(() => { + deps.db.close(); + process.exit(exitCode); + }); + // Fallback: a long-lived SSE connection may block the close callback, so force exit after 1s. + setTimeout(() => process.exit(exitCode), 1000).unref(); +} + +process.on("SIGINT", () => void shutdown("SIGINT")); +process.on("SIGTERM", () => void shutdown("SIGTERM")); + +// Process-level error fallback: once a background +// fire-and-forget promise (title generation, Session drive, etc.) throws, the error +// reaches the process without passing through any catch — persist it first for a +// record, then handle each case according to its nature. +process.on("uncaughtException", (err) => { + console.error(`[server] 未捕获异常: ${err.stack ?? err.message}`); + deps.errors.record({ source: "process", err, code: "uncaught_exception" }); + // From this point the process state can't be trusted (the error was never converged + // by any catch): don't swallow it — wrap up per existing shutdown semantics and exit + // with a nonzero code (equivalent to Node's default crash exit, just with an extra + // persist and graceful wrap-up). + // Must exit even if shutdown itself errors — never let "caught a fatal error" turn + // into "the process limps along in a broken state". + void shutdown("uncaughtException", 1).catch(() => process.exit(1)); +}); +process.on("unhandledRejection", (reason) => { + const err = reason instanceof Error ? reason : new Error(String(reason)); + console.error(`[server] 未处理的 Promise 拒绝: ${err.stack ?? err.message}`); + deps.errors.record({ source: "process", err, code: "unhandled_rejection" }); + // Unlike uncaughtException, this **doesn't** exit: a rejected promise is a localized + // failure of some background task, and the process state isn't compromised; dragging + // down the entire service for it (Node's default behavior) isn't worth it — persist + + // log, then keep serving. +}); diff --git a/packages/server/src/internal/dates.ts b/packages/server/src/internal/dates.ts new file mode 100644 index 0000000..08dafb7 --- /dev/null +++ b/packages/server/src/internal/dates.ts @@ -0,0 +1,20 @@ +/** + * Local date helpers (same convention as core's Trace date directories: local timezone + * yyyy-mm-dd). Shared by the usage_records.date aggregation key and stats windows + * (today / last 7 days / last 30 days). + */ + +/** Format a time as a local `yyyy-mm-dd` (4-digit year, zero-padded 2-digit month/day). */ +export function formatLocalDate(date: Date): string { + const year = date.getFullYear().toString().padStart(4, "0"); + const month = (date.getMonth() + 1).toString().padStart(2, "0"); + const day = date.getDate().toString().padStart(2, "0"); + return `${year}-${month}-${day}`; +} + +/** Subtract N days from a local date (used for the start of the last-7-days / last-30-days windows). */ +export function localDateMinusDays(date: Date, days: number): string { + const d = new Date(date); + d.setDate(d.getDate() - days); + return formatLocalDate(d); +} diff --git a/packages/server/src/runtime/approvals.ts b/packages/server/src/runtime/approvals.ts new file mode 100644 index 0000000..03dd5bb --- /dev/null +++ b/packages/server/src/runtime/approvals.ts @@ -0,0 +1,116 @@ +/** + * Tool call approval: ApproveFn factory + pending + * approval registry. + * + * Approval mode semantics match the CLI (packages/cli/src/approval.ts): + * allow-all auto-approves; deny-all auto-denies; read-only allows read-only tools + * (permission==="r") and routes the rest to manual approval; always-ask routes + * everything to manual approval. Routing to manual approval registers a pending entry + * and pushes an `approval_request` via SSE, suspending until the frontend decides via + * `POST /approvals/:toolCallId`; no timeout — pending approvals are resolved to deny + * when the Task is interrupted (then proceeds through the abort flow). + * + * Every approval decision re-reads the current approval_mode (`getMode` reads the DB), + * so mode changes take effect immediately. + * Docs: /docs/tools § "Approval". + */ +import type { ApprovalMode } from "../api/types.js"; +import type { + ApprovalDecision, + ApproveFn, + OmniMessage, + ToolCallPayload, +} from "@prismshadow/penguin-core"; + +export interface PendingApproval { + toolCall: OmniMessage; + origin?: string[]; +} + +interface PendingEntry extends PendingApproval { + resolve: (decision: ApprovalDecision) => void; +} + +/** Pending approval registry (key = tool_call_id), one per Session runtime. */ +export class ApprovalRegistry { + private readonly pending = new Map(); + + get size(): number { + return this.pending.size; + } + + /** All currently pending approvals (for subscription replay). */ + list(): PendingApproval[] { + return [...this.pending.values()].map(({ toolCall, origin }) => ({ + toolCall, + ...(origin !== undefined ? { origin } : {}), + })); + } + + /** Register and wait for a decision. Re-registering the same id (defensive) resolves the old entry as deny. */ + wait(toolCall: OmniMessage): Promise { + const id = toolCall.payload.tool_call_id; + this.pending.get(id)?.resolve("deny"); + return new Promise((resolve) => { + const entry: PendingEntry = { + toolCall, + ...(toolCall.origin !== undefined ? { origin: toolCall.origin } : {}), + resolve: (decision) => { + this.pending.delete(id); + resolve(decision); + }, + }; + this.pending.set(id, entry); + }); + } + + /** Submit a decision; returns false if not found (already decided/unknown). */ + decide(toolCallId: string, decision: ApprovalDecision): boolean { + const entry = this.pending.get(toolCallId); + if (!entry) return false; + entry.resolve(decision); + return true; + } + + /** Interruption convergence: resolve all pending approvals as deny. */ + denyAll(): void { + for (const entry of [...this.pending.values()]) entry.resolve("deny"); + } +} + +/** + * Build the approve callback: re-reads the approval mode on every call; when routed to + * manual approval, registers a pending entry and suspends after pushing an + * `approval_request` server event via `publishRequest`. + */ +export function makeApprove(args: { + getMode: () => ApprovalMode; + toolPermission: (name: string) => "r" | "rw" | undefined; + registry: ApprovalRegistry; + publishRequest: (pending: PendingApproval) => void; +}): ApproveFn { + const { getMode, toolPermission, registry, publishRequest } = args; + const manual = (toolCall: OmniMessage): Promise => { + const promise = registry.wait(toolCall); + publishRequest({ + toolCall, + ...(toolCall.origin !== undefined ? { origin: toolCall.origin } : {}), + }); + return promise; + }; + return async (toolCall) => { + switch (getMode()) { + case "allow-all": + return "allow"; + case "deny-all": + return "deny"; + case "read-only": + // Auto-approve read-only tools; route read-write/unknown tools to manual approval (matches CLI semantics). + if (toolPermission(toolCall.payload.name) === "r") return "allow"; + return manual(toolCall); + case "always-ask": + default: + return manual(toolCall); + } + }; +} diff --git a/packages/server/src/runtime/channel.ts b/packages/server/src/runtime/channel.ts new file mode 100644 index 0000000..d9b2216 --- /dev/null +++ b/packages/server/src/runtime/channel.ts @@ -0,0 +1,194 @@ +/** + * SSE event channel. + * + * The Session channel and the user channel share this implementation: + * - Event id is an opaque string `-`: epoch is a random short string + * generated when each Channel instance is created, seq is a monotonically increasing + * integer within the channel. epoch necessarily changes when the channel is + * recycled/recreated or the process restarts, so a stale Last-Event-ID always misses + * and falls through to resync — this prevents a silent false-hit event loss when the + * new epoch's event count happens to exceed the old id; + * - A bounded ring buffer (most recent 1000 entries or 2MB, whichever comes first, + * evicting the oldest on overflow) serves replay-on-reconnect via `Last-Event-ID`; + * an evicted/unknown id is handled by the caller sending `resync_required`; + * - Unicast (sendTo) is used for one-off replay at subscribe time (pending approvals / + * resync / hello): it consumes a seq number but doesn't enter the buffer or get + * broadcast — if that subscriber later reconnects with this id, the hit check is + * still safe (seq is monotonic). + * + * This module only handles event numbering / buffering / dispatch, not HTTP — SSE + * output is adapted at the routing layer. + * Docs: /docs/server-api § "Delivery Guarantees". + */ +import { randomUUID } from "node:crypto"; + +/** A numbered channel event; `id` is `-`, `data` is serialized single-line JSON. */ +export interface ChannelEvent { + id: string; + /** SSE event name; omitted (OmniMessage) means no `event:` line. */ + event?: string; + data: string; +} + +export type ChannelListener = (evt: ChannelEvent) => void; + +export interface ChannelOptions { + maxBufferCount?: number; + maxBufferBytes?: number; +} + +const DEFAULT_MAX_COUNT = 1000; +const DEFAULT_MAX_BYTES = 2 * 1024 * 1024; + +/** Buffered entry: seq is stored separately so hit checks never need to parse the string id. */ +interface BufferedEvent { + seq: number; + evt: ChannelEvent; +} + +export class Channel { + /** Channel epoch: generated at instance creation, prefixed onto event ids (necessarily changes after recycle/recreate or restart). */ + readonly epoch: string = randomUUID().slice(0, 8); + private nextSeq = 1; + private buffer: BufferedEvent[] = []; + private bufferBytes = 0; + /** Max seq among evicted events (0 means never evicted): lower bound for hit checks. */ + private lastEvictedSeq = 0; + private readonly listeners = new Set(); + private readonly maxCount: number; + private readonly maxBytes: number; + /** Timestamp of last activity (publish/subscription change), used for idle-reclaim checks. */ + lastActivityMs = Date.now(); + + constructor(opts: ChannelOptions = {}) { + this.maxCount = opts.maxBufferCount ?? DEFAULT_MAX_COUNT; + this.maxBytes = opts.maxBufferBytes ?? DEFAULT_MAX_BYTES; + } + + get subscriberCount(): number { + return this.listeners.size; + } + + /** Broadcast an event: number it, buffer it (evicting the oldest), notify all subscribers. */ + publish(data: unknown, event?: string): ChannelEvent { + const entry = this.makeEvent(data, event); + this.buffer.push(entry); + this.bufferBytes += entry.evt.data.length; + while ( + this.buffer.length > 0 && + (this.buffer.length > this.maxCount || this.bufferBytes > this.maxBytes) + ) { + const evicted = this.buffer.shift()!; + this.bufferBytes -= evicted.evt.data.length; + this.lastEvictedSeq = Math.max(this.lastEvictedSeq, evicted.seq); + } + for (const listener of this.listeners) listener(entry.evt); + return entry.evt; + } + + /** Unicast an event to a single subscriber: consumes a seq but doesn't buffer or broadcast (used for replay at subscribe time). */ + sendTo(listener: ChannelListener, data: unknown, event?: string): ChannelEvent { + const entry = this.makeEvent(data, event); + listener(entry.evt); + return entry.evt; + } + + subscribe(listener: ChannelListener): () => void { + this.listeners.add(listener); + this.lastActivityMs = Date.now(); + return () => { + this.listeners.delete(listener); + this.lastActivityMs = Date.now(); + }; + } + + /** + * Compute replay from a Last-Event-ID (`-`): a mismatched epoch (channel + * recycled/recreated, process restarted, or malformed id) always misses; a matching + * epoch hits the buffer (if no events after that seq have been evicted and the seq was + * indeed assigned by this channel) and returns the buffered events after it; otherwise + * miss (the caller should send `resync_required` first). + */ + replayAfter(lastEventId: string): { hit: boolean; events: ChannelEvent[] } { + const sep = lastEventId.lastIndexOf("-"); + if (sep <= 0) return { hit: false, events: [] }; + const epoch = lastEventId.slice(0, sep); + const seq = Number.parseInt(lastEventId.slice(sep + 1), 10); + if (epoch !== this.epoch || !Number.isInteger(seq) || seq < 0) { + return { hit: false, events: [] }; + } + const hit = seq >= this.lastEvictedSeq && seq < this.nextSeq; + if (!hit) return { hit: false, events: [] }; + return { hit: true, events: this.buffer.filter((e) => e.seq > seq).map((e) => e.evt) }; + } + + private makeEvent(data: unknown, event?: string): BufferedEvent { + this.lastActivityMs = Date.now(); + const serialized = typeof data === "string" ? data : JSON.stringify(data); + const seq = this.nextSeq++; + const evt: ChannelEvent = { id: `${this.epoch}-${seq}`, data: serialized }; + if (event !== undefined) evt.event = event; + return { seq, evt }; + } +} + +const DEFAULT_IDLE_MS = 30 * 60 * 1000; +const SWEEP_INTERVAL_MS = 60 * 1000; + +export interface ChannelHubOptions { + idleMs?: number; + /** + * Active check: keys for which this returns true are excluded from idle reclaim (app + * assembly injects `manager.statusOf(key) !== "idle"`, so a running/compacting + * Session channel is never reclaimed no matter how long since its last publish; a + * user channel key looks like `user:` and is always considered active). + */ + isActive?: (key: string) => boolean; +} + +/** + * Channel collection: lazily created by key (Session id or `user:`); + * a channel whose Session is idle and has had no subscribers for over 30 minutes is + * reclaimed, releasing its buffer as well. + */ +export class ChannelHub { + private readonly channels = new Map(); + private readonly timer: NodeJS.Timeout; + private readonly idleMs: number; + private readonly isActive: (key: string) => boolean; + + constructor(opts: ChannelHubOptions = {}) { + this.idleMs = opts.idleMs ?? DEFAULT_IDLE_MS; + this.isActive = opts.isActive ?? (() => false); + this.timer = setInterval(() => this.sweep(), SWEEP_INTERVAL_MS); + this.timer.unref?.(); + } + + get(key: string): Channel { + let ch = this.channels.get(key); + if (!ch) { + ch = new Channel(); + this.channels.set(key, ch); + } + return ch; + } + + peek(key: string): Channel | undefined { + return this.channels.get(key); + } + + /** Reclaim idle channels (skips active Sessions: no reclaim even without a publish while awaiting approval); `now` is injectable for tests. */ + sweep(now: number = Date.now()): void { + for (const [key, ch] of this.channels) { + if (this.isActive(key)) continue; + if (ch.subscriberCount === 0 && now - ch.lastActivityMs > this.idleMs) { + this.channels.delete(key); + } + } + } + + dispose(): void { + clearInterval(this.timer); + this.channels.clear(); + } +} diff --git a/packages/server/src/runtime/error-recorder.ts b/packages/server/src/runtime/error-recorder.ts new file mode 100644 index 0000000..8cf3085 --- /dev/null +++ b/packages/server/src/runtime/error-recorder.ts @@ -0,0 +1,161 @@ +/** + * Error persistence: errors caught on the server are all + * written to error_records through here, for display on the stats dashboard. Shape + * mirrors usage-recorder — persist only raw facts, leave aggregation to query time. + * + * **The classification (kind) criterion is "does a human need to step in"**, not where + * the error originated: + * + * - `expected`: anticipated by the system, has a defined handling path, part of normal + * operation, no human needed — HTTP business errors (`HttpError`, mostly 4xx); LLM + * `timeout` / `malformed` (the engine already reconnects and retries); tool execution + * `failed` / `timeout` (the error is fed back to the model, and the Agent adjusts on + * its own). + * - `unexpected`: shouldn't happen, usually a bug or a config/environment fault, + * **needs a human** — internal errors converged to 500; process crashes; runtime + * errors escaping from background tasks (Session drive / usage persistence / title + * generation / subagent registration); LLM `failed` (not retryable: auth failure, + * invalid params, etc.). + * - User-initiated actions **are not errors** and are never recorded: request/tool + * `aborted` (user clicked "stop", or denied a tool). + * + * Determination: HTTP sources are inferred automatically from `HttpError` (preserving + * existing behavior); other sources must pass `kind` explicitly at the capture site. + * The frontend highlights unexpected by default; expected is still recorded without + * losing information. + * + * Sources cover HTTP, Session drive, LLM requests, Environment (tool execution), usage + * persistence, title generation, subagent registration, and process-level fallback; + * among these, `llm` / `environment` errors are not expressed via throw (core converges + * them into the message stream instead), and are fished out by stream-error-watcher from + * the Session output stream. + * + * **This recorder never throws**: it's hooked onto app.onError, and throwing from + * within it would turn error handling into infinite recursion; if persistence itself + * fails (disk full / DB already closed, etc.), it's fine to drop that one record. + * + * **Short-window dedup (DEDUP_WINDOW_MS)**: error storms are the norm — someone scanning + * the API produces a wall of 404s, or a tool fails repeatedly in a loop. Persisting each + * one both write-amplifies and floods the table, and makes the dashboard's "most recent + * 20" all the same error. So the same `(source, code, Project)` is persisted at most + * once per window; repeats within the window are **dropped outright** (not persisted); + * only an actual persist refreshes the timestamp, so a sustained storm leaves a steady + * one record per window instead of being suppressed indefinitely. + * **Tradeoff**: aggregate counts therefore **underestimate** — a storm of the same error + * only counts once, check the logs for true frequency; in exchange, a single error storm + * doesn't drown out error_records or the stats dashboard. The second line of defense is + * ErrorsRepo's capacity cap. The dedup table (lastSeen) must stay bounded: past + * DEDUP_KEYS_MAX, expired entries are cleared first, and if still over the limit the + * whole table is cleared — better to miss some dedup than let it grow unbounded across + * different error codes. + */ +import { formatLocalDate } from "../internal/dates.js"; +import { HttpError } from "../http/errors.js"; +import type { ErrorsRepo } from "../db/repos/errors.js"; + +/** Capture-site source (maps one-to-one to error_records.source). */ +export type ErrorSource = + | "http" + | "session" + | "llm" + | "environment" + | "usage" + | "title" + | "subagent" + | "process" + | "schedule"; + +/** Error classification: see file header — the criterion is "does a human need to step in". */ +export type ErrorKind = "expected" | "unexpected"; + +/** Attribution context (all optional: the login endpoint has no Project, and process-level fallback has no request at all). */ +export interface ErrorContext { + projectId?: string; + agentId?: string; + sessionId?: string; +} + +export interface ErrorRecordArgs { + source: ErrorSource; + /** The caught error (unknown: the value caught may not be an Error; failures from the message stream pass the reason text directly). */ + err: unknown; + ctx?: ErrorContext; + /** Semantic code (required for non-HTTP sources, e.g. session_run_failed); defaults to HttpError.code. */ + code?: string; + /** HTTP status code; leave empty for non-HTTP sources. */ + status?: number; + /** Explicit classification (see file header); defaults to inferring from `HttpError` — HTTP sources rely on this, other sources should pass it explicitly. */ + kind?: ErrorKind; +} + +/** Message truncation length (keep only a readable summary; the full stack is still logged). */ +export const MESSAGE_MAX = 500; + +/** Short-window dedup window: the same (source, code, Project) is persisted at most once per window (see the file header's tradeoff). */ +export const DEDUP_WINDOW_MS = 2000; + +/** Cap on dedup table keys (bounded; over the limit, expired entries are cleared first, and if still over, the whole table is cleared). */ +export const DEDUP_KEYS_MAX = 1000; + +function messageOf(err: unknown): string { + const raw = err instanceof Error ? err.message : String(err); + return raw.length > MESSAGE_MAX ? raw.slice(0, MESSAGE_MAX) : raw; +} + +export class ErrorRecorder { + /** Dedup table: `source \0 code \0 projectId` → timestamp of the last **persist** (see file header). */ + private readonly lastSeen = new Map(); + + constructor( + private readonly errors: ErrorsRepo, + private readonly now: () => Date = () => new Date(), + ) {} + + /** Record an error (synchronous, fails silently; same-window duplicates are dropped outright, see file header). */ + record(args: ErrorRecordArgs): void { + try { + const http = args.err instanceof HttpError ? args.err : null; + const now = this.now(); + const projectId = args.ctx?.projectId ?? null; + const code = args.code ?? http?.code ?? "internal"; + // Short-window dedup: coarse-grained to "same kind of error for the same Project"; repeats within the window aren't persisted. + if (this.deduped(`${args.source}\0${code}\0${projectId ?? ""}`, now.getTime())) return; + this.errors.insert({ + ts: now.toISOString(), + date: formatLocalDate(now), + projectId, + agentId: args.ctx?.agentId ?? null, + sessionId: args.ctx?.sessionId ?? null, + source: args.source, + // Explicit classification takes priority; otherwise infer from HttpError (business error = expected, else unexpected). + kind: args.kind ?? (http ? "expected" : "unexpected"), + code, + // Unexpected errors from HTTP sources are converged to 500 externally (matches handleError's response). + status: args.status ?? http?.status ?? (args.source === "http" ? 500 : null), + message: messageOf(args.err), + }); + } catch { + // See file header: if the recorder itself errors, dropping this one record is the only option — never rethrow. + } + } + + /** true if a same-kind error was already recorded within the window (drop it); otherwise register this persist timestamp and keep the dedup table bounded. */ + private deduped(key: string, nowMs: number): boolean { + const last = this.lastSeen.get(key); + if (last !== undefined && nowMs - last < DEDUP_WINDOW_MS) return true; + this.lastSeen.set(key, nowMs); + if (this.lastSeen.size > DEDUP_KEYS_MAX) this.evict(nowMs); + return false; + } + + /** Keep the dedup table bounded (see file header): clear expired entries first; if still over the limit (hundreds/thousands of distinct error codes erupting at once), clear it entirely. */ + private evict(nowMs: number): void { + for (const [key, at] of this.lastSeen) { + if (nowMs - at >= DEDUP_WINDOW_MS) this.lastSeen.delete(key); + } + if (this.lastSeen.size > DEDUP_KEYS_MAX) this.lastSeen.clear(); + } +} + +/** Minimal dependency a capture site needs on the recorder (tests inject a fake; structurally matches SessionManager's UsageRecorderLike). */ +export type ErrorSink = Pick; diff --git a/packages/server/src/runtime/schedule-file.ts b/packages/server/src/runtime/schedule-file.ts new file mode 100644 index 0000000..18463ca --- /dev/null +++ b/packages/server/src/runtime/schedule-file.ts @@ -0,0 +1,214 @@ +/** + * Schedule file parsing, validation, and trigger-time computation. + * + * `agent_state/schedule/.toml` is declarative intent; the system never writes it + * back. This module does pure parsing and pure time math only: an invalid file returns + * an error (the scheduler skips it and records the error); runtime state (fired / + * missed / disabled) doesn't live here — it belongs to SQLite (db/repos/schedules.ts). + * Docs: /docs/configuration § "Schedules". + */ +import { parse as parseToml } from "smol-toml"; + +/** `period` lower bound: below 5 minutes is treated as an invalid file (guards against runaway high-frequency tasks). */ +export const MIN_PERIOD_MS = 5 * 60_000; + +/** A parsed schedule definition (the filename minus `.toml` is its identity). */ +export interface ScheduleDefinition { + name: string; + /** The Prompt to send (required). */ + prompt: string; + /** Enabled switch; disabled by default. */ + enabled: boolean; + /** Original text of the first trigger time (for API echo, preserving the written form). */ + startAt: string; + /** First trigger time (epoch ms). */ + startAtMs: number; + /** Original text of the end time. */ + endAt?: string; + /** Original text of the trigger period (e.g. `30m`, for API echo); undefined means a one-shot task. */ + period?: string; + /** Trigger period (ms); undefined means a one-shot task. */ + periodMs?: number; + /** End time (epoch ms); no more triggers once past it. */ + endAtMs?: number; + /** The target Session to bind to; defaults to creating a new Session each time. */ + sessionId?: string; + /** Workspace for new-Session mode (same semantics as manually starting a session; auto-creates a temp directory if unspecified). */ + workspace?: string; + /** Model for new-Session mode (upstream id, paired with provider; defaults to the Project's default reference). */ + modelId?: string; + /** + * Vendor grouping for `model_id` (paired reference); when omitted, resolved per + * resolveModelRef semantics — whether the reference is resolvable is validated by the + * caller against config at reconciliation/save time (this module does pure parsing + * and never touches config). + */ + provider?: string; +} + +export type ScheduleParseResult = + { ok: true; def: ScheduleDefinition } | { ok: false; error: string }; + +/** Parse a fixed interval in `30m` / `12h` / `7d` form; returns null if invalid. */ +export function parsePeriod(raw: string): number | null { + const m = /^(\d+)([mhd])$/.exec(raw.trim()); + if (!m) return null; + const n = Number(m[1]); + if (!Number.isInteger(n) || n <= 0) return null; + const unit = m[2] === "m" ? 60_000 : m[2] === "h" ? 3_600_000 : 86_400_000; + return n * unit; +} + +/** Parse an ISO 8601 instant into epoch ms plus the original text for echo; returns null if invalid (smol-toml's date values are also accepted). */ +function parseInstant(value: unknown): { ms: number; raw: string } | null { + if (value instanceof Date) { + const ms = value.getTime(); + return Number.isNaN(ms) ? null : { ms, raw: value.toISOString() }; + } + if (typeof value !== "string") return null; + const ms = Date.parse(value); + return Number.isNaN(ms) ? null : { ms, raw: value }; +} + +/** + * Parse and validate a schedule file. A field with the wrong type invalidates the whole + * file (the baseline for hand-edit tolerance is to never let bad config reach the + * scheduler); unknown keys are ignored (forward compatibility). + */ +export function parseScheduleFile(name: string, raw: string): ScheduleParseResult { + let parsed: unknown; + try { + parsed = parseToml(raw); + } catch (err) { + return { + ok: false, + error: `TOML 解析失败:${err instanceof Error ? err.message : String(err)}`, + }; + } + if (parsed === null || typeof parsed !== "object") + return { ok: false, error: "内容不是 TOML 表" }; + const t = parsed as Record; + + const prompt = t["prompt"]; + if (typeof prompt !== "string" || prompt.trim() === "") { + return { ok: false, error: "缺少必填字段 prompt" }; + } + const enabled = t["enabled"] === undefined ? false : t["enabled"]; + if (typeof enabled !== "boolean") return { ok: false, error: "enabled 必须是布尔值" }; + + const startAt = parseInstant(t["start_at"]); + if (startAt === null) return { ok: false, error: "start_at 缺失或不是合法的 ISO 8601 时刻" }; + + let period: string | undefined; + let periodMs: number | undefined; + if (t["period"] !== undefined) { + if (typeof t["period"] !== "string") return { ok: false, error: "period 必须是字符串" }; + const ms = parsePeriod(t["period"]); + if (ms === null) return { ok: false, error: "period 必须形如 30m / 12h / 7d" }; + if (ms < MIN_PERIOD_MS) return { ok: false, error: "period 低于下限 5m" }; + period = t["period"].trim(); + periodMs = ms; + } + + let endAt: { ms: number; raw: string } | undefined; + if (t["end_at"] !== undefined) { + const parsedEnd = parseInstant(t["end_at"]); + if (parsedEnd === null) return { ok: false, error: "end_at 不是合法的 ISO 8601 时刻" }; + if (parsedEnd.ms <= startAt.ms) return { ok: false, error: "end_at 必须晚于 start_at" }; + endAt = parsedEnd; + } + + let sessionId: string | undefined; + if (t["session_id"] !== undefined) { + if (typeof t["session_id"] !== "string" || t["session_id"] === "") { + return { ok: false, error: "session_id 必须是非空字符串" }; + } + sessionId = t["session_id"]; + } + let workspace: string | undefined; + if (t["workspace"] !== undefined) { + if (typeof t["workspace"] !== "string" || t["workspace"] === "") { + return { ok: false, error: "workspace 必须是非空字符串" }; + } + workspace = t["workspace"]; + } + let modelId: string | undefined; + if (t["model_id"] !== undefined) { + if (typeof t["model_id"] !== "string" || t["model_id"] === "") { + return { ok: false, error: "model_id 必须是非空字符串" }; + } + modelId = t["model_id"]; + } + let provider: string | undefined; + if (t["provider"] !== undefined) { + if (typeof t["provider"] !== "string" || t["provider"] === "") { + return { ok: false, error: "provider 必须是非空字符串" }; + } + provider = t["provider"]; + } + if (provider !== undefined && modelId === undefined) { + return { ok: false, error: "provider 仅与 model_id 成对使用(模型引用须成对给出)" }; + } + if ( + sessionId !== undefined && + (workspace !== undefined || modelId !== undefined || provider !== undefined) + ) { + return { + ok: false, + error: "目标二选一:workspace 与 provider / model_id 仅用于新建 Session 模式", + }; + } + + return { + ok: true, + def: { + name, + prompt, + enabled, + startAt: startAt.raw, + startAtMs: startAt.ms, + ...(period !== undefined ? { period } : {}), + ...(periodMs !== undefined ? { periodMs } : {}), + ...(endAt !== undefined ? { endAt: endAt.raw, endAtMs: endAt.ms } : {}), + ...(sessionId !== undefined ? { sessionId } : {}), + ...(workspace !== undefined ? { workspace } : {}), + ...(modelId !== undefined ? { modelId } : {}), + ...(provider !== undefined ? { provider } : {}), + }, + }; +} + +/** + * Step from `start_at` by `period` and return the most recent scheduled time not later + * than `nowMs`; null if `start_at` hasn't been reached yet. A one-shot task's only slot + * is `start_at` itself. + */ +export function latestSlotAt(def: ScheduleDefinition, nowMs: number): number | null { + if (nowMs < def.startAtMs) return null; + if (def.periodMs === undefined) return def.startAtMs; + const k = Math.floor((nowMs - def.startAtMs) / def.periodMs); + return def.startAtMs + k * def.periodMs; +} + +/** Whether a scheduled slot still falls within the `[start_at, end_at]` window (always true if there's no end_at). */ +export function slotInWindow(def: ScheduleDefinition, slotMs: number): boolean { + return def.endAtMs === undefined || slotMs <= def.endAtMs; +} + +/** + * The next scheduled time strictly after `nowMs` (used to display "next trigger"); + * for a one-shot task this only has a value while start_at hasn't been reached, and + * returns null once past end_at. + */ +export function nextSlotAfter(def: ScheduleDefinition, nowMs: number): number | null { + let next: number; + if (nowMs < def.startAtMs) { + next = def.startAtMs; + } else if (def.periodMs === undefined) { + return null; + } else { + const k = Math.floor((nowMs - def.startAtMs) / def.periodMs) + 1; + next = def.startAtMs + k * def.periodMs; + } + return slotInWindow(def, next) ? next : null; +} diff --git a/packages/server/src/runtime/schedule-store.ts b/packages/server/src/runtime/schedule-store.ts new file mode 100644 index 0000000..46b58cd --- /dev/null +++ b/packages/server/src/runtime/schedule-store.ts @@ -0,0 +1,140 @@ +/** + * Schedule file access: `agent_state/schedule/.toml`, where + * the filename (a semantic name) is the identity. Reads are fault-tolerant (an invalid + * file is recorded as an error and skipped by the caller); writes only go through the + * API routes (the system never rewrites existing file content — PUT is a full-file + * replacement expressing user intent). + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { stringify as stringifyToml } from "smol-toml"; +import { loadProjectConfig, resolveModelRef, scheduleDir } from "@prismshadow/penguin-core"; +import type { ScheduleDefinition } from "./schedule-file.js"; +import { parseScheduleFile, type ScheduleParseResult } from "./schedule-file.js"; + +export interface ScheduleFileEntry { + name: string; + raw: string; + parsed: ScheduleParseResult; +} + +/** List all schedule files for this Agent (a missing directory is treated as empty). */ +export async function listScheduleFiles( + root: string, + projectId: string, + agentId: string, +): Promise { + const dir = scheduleDir(root, projectId, agentId); + let names: string[]; + try { + names = await fs.readdir(dir); + } catch { + return []; + } + const entries: ScheduleFileEntry[] = []; + for (const file of names.sort()) { + if (!file.endsWith(".toml")) continue; + const name = file.slice(0, -".toml".length); + let raw: string; + try { + raw = await fs.readFile(path.join(dir, file), "utf8"); + } catch { + continue; // Deleted during reconciliation: revisit next round. + } + entries.push({ name, raw, parsed: parseScheduleFile(name, raw) }); + } + return entries; +} + +export async function readScheduleFile( + root: string, + projectId: string, + agentId: string, + name: string, +): Promise { + const file = path.join(scheduleDir(root, projectId, agentId), `${name}.toml`); + try { + const raw = await fs.readFile(file, "utf8"); + return { name, raw, parsed: parseScheduleFile(name, raw) }; + } catch { + return null; + } +} + +/** Serialize API fields into file content (validation uniformly goes through parseScheduleFile, avoiding two sets of rules). */ +export function serializeSchedule(fields: { + prompt: string; + enabled: boolean; + startAt: string; + period?: string; + endAt?: string; + sessionId?: string; + workspace?: string; + modelId?: string; + provider?: string; +}): string { + const table: Record = { + prompt: fields.prompt, + enabled: fields.enabled, + start_at: fields.startAt, + ...(fields.period !== undefined ? { period: fields.period } : {}), + ...(fields.endAt !== undefined ? { end_at: fields.endAt } : {}), + ...(fields.sessionId !== undefined ? { session_id: fields.sessionId } : {}), + ...(fields.workspace !== undefined ? { workspace: fields.workspace } : {}), + ...(fields.provider !== undefined ? { provider: fields.provider } : {}), + ...(fields.modelId !== undefined ? { model_id: fields.modelId } : {}), + }; + return `${stringifyToml(table)}\n`; +} + +/** + * Resolvability check for a schedule's model reference (shared by save and + * reconciliation): when the definition has `model_id`, it's + * resolved against Project config per resolveModelRef semantics — omitting provider is + * only resolvable when model_id matches exactly one entry globally; zero hits or + * ambiguity means unresolvable. Returns an error message (unresolvable / config read + * failure), or null if resolvable (or no model reference at all). + */ +export async function validateScheduleModelRef( + root: string, + projectId: string, + def: Pick, +): Promise { + if (def.modelId === undefined) return null; + try { + const cfg = await loadProjectConfig(root, projectId); + resolveModelRef(cfg, def.modelId, def.provider); + return null; + } catch (err) { + return err instanceof Error ? err.message : String(err); + } +} + +/** Write a schedule file to disk (full-file replacement for POST/PUT). */ +export async function writeScheduleFile( + root: string, + projectId: string, + agentId: string, + name: string, + raw: string, +): Promise { + const dir = scheduleDir(root, projectId, agentId); + await fs.mkdir(dir, { recursive: true }); + await fs.writeFile(path.join(dir, `${name}.toml`), raw, "utf8"); +} + +/** Delete a schedule file; returns false if it doesn't exist. */ +export async function deleteScheduleFile( + root: string, + projectId: string, + agentId: string, + name: string, +): Promise { + const file = path.join(scheduleDir(root, projectId, agentId), `${name}.toml`); + try { + await fs.unlink(file); + return true; + } catch { + return false; + } +} diff --git a/packages/server/src/runtime/scheduler.ts b/packages/server/src/runtime/scheduler.ts new file mode 100644 index 0000000..1a7e75b Binary files /dev/null and b/packages/server/src/runtime/scheduler.ts differ diff --git a/packages/server/src/runtime/session-manager.ts b/packages/server/src/runtime/session-manager.ts new file mode 100644 index 0000000..3d85309 --- /dev/null +++ b/packages/server/src/runtime/session-manager.ts @@ -0,0 +1,841 @@ +/** + * Active Session runtime. + * + * Responsibilities: + * - get-or-resume-or-heal: use it directly on an active-table hit; with a Trace, + * recover via `agent.resumeSession`; a stale Session that was created but never run + * and survived a process restart (no Trace) **self-heals** — recreated via + * createSession using the index row's workspace/modelId, yielding a new session_id + * and updating the index's primary key; the Task response body always returns the + * current actual id; + * - Per-Session mutual exclusion: only one Task/compaction may be in progress at a + * time; + * - run/compact drive: consumes the output stream in the background, publishing each + * message to the SSE channel and handing it to usage-recorder for persistence; + * on completion (including errors) resets to idle and pushes a `task_state` server + * event; + * - Approval registration and interrupt convergence: each approval decision re-reads + * approval_mode from the DB (takes effect immediately); an interrupt first + * converges pending approvals to deny, then aborts. + * + * The underlying implementation of get-or-resume-or-heal is injected via + * `SessionLoader`: production uses the core SDK (createCoreSessionLoader), tests inject + * a fake Session (issuing no real LLM requests). + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { + createAgent, + findLatestTraceFile, + isSessionMeta, + tracesDir, +} from "@prismshadow/penguin-core"; +import type { + ApproveFn, + CompactAvailability, + OmniMessage, + SessionMetaPayload, + SessionTitleResult, + TextPayload, +} from "@prismshadow/penguin-core"; +import type { ServerEvent, SessionStatus } from "../api/types.js"; +import { HttpError, isMissingCredential, modelCredentialMissing } from "../http/errors.js"; +import type { SessionRow, SessionsRepo } from "../db/repos/sessions.js"; +import { ApprovalRegistry, makeApprove } from "./approvals.js"; +import type { PendingApproval } from "./approvals.js"; +import type { ChannelHub } from "./channel.js"; +import type { ErrorSink } from "./error-recorder.js"; +import { StreamErrorWatcher } from "./stream-error-watcher.js"; +import type { TitleNotifier } from "./title-generator.js"; +import type { UsageContext } from "./usage-recorder.js"; + +/** 409 for when there's nothing to compact: give the specific reason rather than a one-size-fits-none message. */ +function compactUnavailable(why: Exclude): HttpError { + const messages: Record = { + unsupported: "该 Agent 未配置上下文压缩能力。", + empty: "当前上下文没有可压缩的内容(尚无已完成的对话轮次)。", + just_compacted: "上下文刚压缩过,此后还没有新的对话,无需再次压缩。", + }; + return new HttpError(409, "nothing_to_compact", messages[why]); +} + +/** Minimal interface for a runtime Session (satisfied by core Session; tests may inject a fake implementation). */ +export interface RuntimeSession { + readonly sessionId: string; + run( + newMessages: OmniMessage[], + opts: { approve: ApproveFn; signal: AbortSignal }, + ): AsyncGenerator; + compact(opts: { signal: AbortSignal }): AsyncGenerator; + /** Whether compaction is possible and why; when not ok, compact() yields no messages (see core ContextEngine.compactability). */ + compactability(): CompactAvailability; + toolPermission(name: string): "r" | "rw" | undefined; + /** + * Out-of-band one-shot request for title generation (core `Session.generateTitle`, + * writes no history/Trace). Material defaults to what the Session collects itself + * (the first Task's text gathered during run); `material` overrides this for + * subagents. + */ + generateTitle(args?: { + material?: { userText: string; assistantText: string }; + signal?: AbortSignal; + }): Promise; +} + +/** The underlying loader behind get-or-resume-or-heal. */ +export interface SessionLoader { + /** + * Load a runtime Session from an index row: recover (with a Trace) or self-heal + * rebuild (no Trace, session_id will change). Throws HttpError(409) for unrecoverable + * cases such as a missing Workspace. + */ + load(row: SessionRow): Promise; +} + +/** Production loader: the core SDK's resumeSession / createSession. */ +export function createCoreSessionLoader(root: string): SessionLoader { + return { + async load(row: SessionRow): Promise { + const agent = await createAgent({ + root, + projectId: row.projectId, + agentId: row.agentId, + }); + const located = await findLatestTraceFile( + tracesDir(root, row.projectId, row.agentId), + row.sessionId, + ); + if (located) { + // With a Trace: rebuild via "Session Recovery" (history injected via setHistory, + // carrying over any residual state). + // core's recognizable recovery failures (Workspace deleted / Model removed from + // config / Trace missing session_meta, etc.) are converged to 409, preserving + // the original message rather than bubbling up as 500. + try { + return await agent.resumeSession({ sessionId: row.sessionId }); + } catch (err) { + // The credential key was deleted after the Session was created: only caught + // here at recovery time; give the same actionable message. + if (isMissingCredential(err)) throw modelCredentialMissing(row.modelId); + throw toUnrecoverableError(err); + } + } + // No Trace (created but never run, and the process has restarted since): self-heal + // rebuild. A missing Workspace → 409. + try { + const stat = await fs.stat(row.workspace); + if (!stat.isDirectory()) throw new Error("not a directory"); + } catch { + throw new HttpError( + 409, + "workspace_missing", + `该 Session 的 Workspace 已不存在:${row.workspace},无法继续。请新建 Session。`, + ); + } + try { + return await agent.createSession({ + workspaceDir: row.workspace, + modelId: row.modelId, + provider: row.provider, + }); + } catch (err) { + if (isMissingCredential(err)) throw modelCredentialMissing(row.modelId); + throw toUnrecoverableError(err); + } + }, + }; +} + +/** A plain Error thrown by core recovery/self-heal rebuild → 409 (preserving the original, actionable message). */ +function toUnrecoverableError(err: unknown): HttpError { + if (err instanceof HttpError) return err; + return new HttpError( + 409, + "session_unrecoverable", + err instanceof Error ? err.message : String(err), + ); +} + +export interface UsageRecorderLike { + record(ctx: UsageContext, msg: OmniMessage): Promise; +} + +export interface SessionManagerDeps { + sessions: SessionsRepo; + channels: ChannelHub; + loader: SessionLoader; + recorder: UsageRecorderLike; + /** Automatic Session title generation (optional: not injected in tests or when disabled). */ + titles?: TitleNotifier; + /** Error persistence (optional: without it, only logs — same as before this was wired up). */ + errors?: ErrorSink; + log?: (line: string) => void; +} + +/** Active-table entry: a loaded runtime Session plus its running state. */ +interface RuntimeEntry { + sessionId: string; + projectId: string; + agentId: string; + /** Vendor grouping for the Session's model (paired with modelId to form a model reference). */ + provider: string; + modelId: string; + session: RuntimeSession; + status: SessionStatus; + approvals: ApprovalRegistry; + abort: AbortController | null; + /** The in-flight drive Promise (awaited during graceful shutdown). */ + running: Promise | null; + /** Timestamp of last activity (refreshed on load / status flip / drive completion), used for idle-eviction checks. */ + lastActivityMs: number; +} + +/** Active-table idle eviction: same convention as the SSE channel (an idle entry with no activity for 30 minutes releases its memory). */ +const ENTRY_IDLE_MS = 30 * 60 * 1000; +const ENTRY_SWEEP_INTERVAL_MS = 60 * 1000; + +/** Cap on collected model text for title material (accumulation stops beyond this; the generator side also truncates further). */ +const TITLE_EXCERPT_LIMIT = 4000; + +/** Composite Agent key (used as a Set key, avoiding projectId/agentId concatenation ambiguity). */ +function agentKey(projectId: string, agentId: string): string { + return `${projectId}\0${agentId}`; +} + +/** If msg is a run_subagent tool call carrying a `prompt`, return its id and prompt (for use as the subagent's title); otherwise null. */ +function runSubagentCall(msg: OmniMessage): { toolCallId: string; prompt: string } | null { + const p = msg.payload as { + type?: string; + name?: string; + arguments?: string; + tool_call_id?: string; + }; + if (msg.type !== "model_msg" || p.type !== "tool_call" || p.name !== "run_subagent") return null; + if (typeof p.arguments !== "string" || typeof p.tool_call_id !== "string") return null; + try { + const args = JSON.parse(p.arguments) as { prompt?: unknown }; + if (typeof args.prompt !== "string" || !args.prompt.trim()) return null; + return { toolCallId: p.tool_call_id, prompt: args.prompt }; + } catch { + return null; // Arguments were truncated/malformed: this call is doomed, no subagent will result + } +} + +/** The denied tool_call_id (approval_decision with decision ≠ allow); otherwise null. */ +function deniedToolCallId(msg: OmniMessage): string | null { + const p = msg.payload as { type?: string; decision?: string; tool_call_id?: string }; + if (msg.type !== "event_msg" || p.type !== "approval_decision") return null; + if (p.decision === "allow" || typeof p.tool_call_id !== "string") return null; + return p.tool_call_id; +} + +/** The tool_call_id of a parent-level tool call that has settled (a complete tool_call_output); otherwise null. */ +function settledToolCallId(msg: OmniMessage): string | null { + const p = msg.payload as { type?: string; tool_call_id?: string }; + if (msg.type !== "model_msg" || p.type !== "tool_call_output") return null; + return typeof p.tool_call_id === "string" ? p.tool_call_id : null; +} + +/** A subagent registered during this run, plus its title material. */ +interface ChildSession { + sessionId: string; + agentId: string; + modelId: string; + /** The prompt of the run_subagent call that spawned it (user material for title generation, and the fallback title). */ + prompt: string; + /** The model text the subagent itself produced (assistant material for title generation). */ + assistantExcerpt: string; +} + +/** Predicate for a plain-text message on the main session (no origin): title material is drawn only from user/model text. */ +function isPlainText(role: "user" | "assistant") { + return (msg: OmniMessage): msg is OmniMessage => { + const payload = msg.payload as { type?: string; role?: string }; + return ( + msg.type === "model_msg" && + payload.type === "text" && + payload.role === role && + (!msg.origin || msg.origin.length === 0) + ); + }; +} + +/** For a nested message, the owning Session (end of the origin chain) and text of the model reply; null if it isn't model text. */ +function nestedAssistantText(msg: OmniMessage): { sessionId: string; text: string } | null { + const p = msg.payload as { type?: string; role?: string; text?: string }; + if (msg.type !== "model_msg" || p.type !== "text" || p.role !== "assistant") return null; + if (!msg.origin || msg.origin.length === 0 || typeof p.text !== "string") return null; + return { sessionId: msg.origin[msg.origin.length - 1]!, text: p.text }; +} + +export class SessionManager { + private readonly entries = new Map(); + /** Per-Session mutex (serializes get-or-load and status flips); auto-cleaned once the chain drains. */ + private readonly locks = new Map>(); + private readonly log: (line: string) => void; + /** Graceful-shutdown flag: once set, new Tasks/compactions are rejected (503). */ + private closed = false; + /** Agents currently being deleted (key = agentKey): new Tasks/compactions are always rejected with 409 during this window. */ + private readonly deletingAgents = new Set(); + /** Sessions currently being deleted (guards against the entry/Trace file being rebuilt and reviving it inside the deletion race window). */ + private readonly deletingSessions = new Set(); + private readonly sweepTimer: NodeJS.Timeout; + + constructor(private readonly deps: SessionManagerDeps) { + this.log = deps.log ?? ((line) => console.error(line)); + this.sweepTimer = setInterval(() => this.sweepIdle(), ENTRY_SWEEP_INTERVAL_MS); + this.sweepTimer.unref?.(); + } + + // —— Query surface (used by Session listing / Agent active-count / SSE subscription replay) —— + + statusOf(sessionId: string): SessionStatus { + return this.entries.get(sessionId)?.status ?? "idle"; + } + + pendingApprovalCount(sessionId: string): number { + return this.entries.get(sessionId)?.approvals.size ?? 0; + } + + pendingApprovals(sessionId: string): PendingApproval[] { + return this.entries.get(sessionId)?.approvals.list() ?? []; + } + + /** Number of Sessions for this Agent that are currently running / compacting. */ + activeCountForAgent(projectId: string, agentId: string): number { + let n = 0; + for (const e of this.entries.values()) { + if (e.projectId === projectId && e.agentId === agentId && e.status !== "idle") n++; + } + return n; + } + + /** Add a newly created Session to the active table (status idle), avoiding a redundant load on the next Task. */ + adopt(row: SessionRow, session: RuntimeSession): void { + this.entries.set(row.sessionId, { + sessionId: row.sessionId, + projectId: row.projectId, + agentId: row.agentId, + provider: row.provider, + modelId: row.modelId, + session, + status: "idle", + approvals: new ApprovalRegistry(), + abort: null, + running: null, + lastActivityMs: Date.now(), + }); + } + + // —— Task / compaction drive —— + + /** + * Start a Task: get-or-load → 409 + * mutual-exclusion check → publish the input messages first → drive run in the + * background. Returns the current actual session_id (the new id after self-heal). + */ + async startTask(sessionId: string, input: OmniMessage[]): Promise<{ sessionId: string }> { + return this.withLock(sessionId, async () => { + this.assertOpen(); + this.assertAgentNotDeleting(sessionId); + this.assertSessionNotDeleting(sessionId); + const entry = await this.ensureEntry(sessionId); + this.assertIdle(entry); + const channel = this.deps.channels.get(entry.sessionId); + const ac = new AbortController(); + entry.status = "running"; + entry.abort = ac; + entry.lastActivityMs = Date.now(); + // Publish the input messages first (visible to other subscribers; the Trace is + // persisted by the SDK), then flip the running status. + for (const msg of input) channel.publish(msg); + this.publishState(entry, "running"); + + const approve = makeApprove({ + // Re-reads approval_mode from the DB on every decision (a PATCH takes effect immediately). + getMode: () => this.deps.sessions.findById(entry.sessionId)?.approvalMode ?? "always-ask", + toolPermission: (name) => entry.session.toolPermission(name), + registry: entry.approvals, + publishRequest: (pending) => + this.publishEvent(entry, { + type: "approval_request", + toolCall: pending.toolCall, + ...(pending.origin !== undefined ? { origin: pending.origin } : {}), + }), + }); + const gen = entry.session.run(input, { approve, signal: ac.signal }); + // Title material is collected by the core Session itself during run; here we only + // keep this call's input user text, used both as the "material present → attempt + // generation" criterion and as the fallback title source if the LLM call fails. + const userExcerpt = input + .filter(isPlainText("user")) + .map((m) => m.payload.text) + .join("\n"); + entry.running = this.drive(entry, gen, { userExcerpt }); + return { sessionId: entry.sessionId }; + }); + } + + /** Manually compact the context: 409 if already running; compaction output also flows into the SSE channel. */ + async startCompact(sessionId: string): Promise<{ sessionId: string }> { + return this.withLock(sessionId, async () => { + this.assertOpen(); + this.assertAgentNotDeleting(sessionId); + this.assertSessionNotDeleting(sessionId); + const entry = await this.ensureEntry(sessionId); + this.assertIdle(entry); + // When there's nothing to compact, core's compact() yields no messages at all: we + // can't just return 202 and walk away, or the frontend would wait forever for a + // compaction banner that never comes (this is exactly the "/compact does nothing + // after an interrupt" complaint). Reject explicitly, and **say why** clearly — + // "just compacted" and "haven't talked yet" share the same internal state + // (sessionTurns === 0), but are two completely different messages to the user: + // telling someone who just compacted that there's "no completed conversation turn + // yet" tells them nothing. + const why = entry.session.compactability(); + if (why !== "ok") throw compactUnavailable(why); + const ac = new AbortController(); + entry.status = "compacting"; + entry.abort = ac; + entry.lastActivityMs = Date.now(); + this.publishState(entry, "compacting"); + const gen = entry.session.compact({ signal: ac.signal }); + entry.running = this.drive(entry, gen); + return { sessionId: entry.sessionId }; + }); + } + + /** Submit an approval decision; returns false if the pending approval doesn't exist (already decided/unknown). */ + decideApproval(sessionId: string, toolCallId: string, decision: "allow" | "deny"): boolean { + const entry = this.entries.get(sessionId); + if (!entry) return false; + return entry.approvals.decide(toolCallId, decision); + } + + /** + * Interrupt the current Task/compaction: pending approvals converge to deny first, + * then the AbortSignal fires. Returns false if nothing is in progress (the route + * treats this as a 204 no-op). + */ + abortTask(sessionId: string): boolean { + const entry = this.entries.get(sessionId); + if (!entry || !entry.abort) return false; + entry.approvals.denyAll(); + entry.abort.abort(); + return true; + } + + /** + * Before deleting a Project, converge all its active runs and clear them out of the + * active table. Returns the in-flight drive Promises of the affected entries: the + * caller (deleteProject) should await them before removing the directory, so that + * interrupt-cleanup Trace writes don't recreate the directory after deletion. + */ + abortProject(projectId: string): Promise[] { + const runnings: Promise[] = []; + for (const [key, entry] of [...this.entries]) { + if (entry.projectId !== projectId) continue; + entry.approvals.denyAll(); + entry.abort?.abort(); + if (entry.running) runnings.push(entry.running); + this.entries.delete(key); + } + return runnings; + } + + /** + * Before deleting an Agent, converge all its active runs and clear them out of the + * active table (same semantics as abortProject). Also marks this Agent as "being + * deleted": new Tasks/compactions entering during the deletion process are always + * rejected with 409 (assertAgentNotDeleting), closing the race window where a new + * task recreates the directory and revives an already-deleted Agent between the + * abortAgent snapshot and the directory removal. The caller must call + * endAgentDeletion once deletion finishes (success or failure). + */ + beginAgentDeletion(projectId: string, agentId: string): Promise[] { + this.deletingAgents.add(agentKey(projectId, agentId)); + const runnings: Promise[] = []; + for (const [key, entry] of [...this.entries]) { + if (entry.projectId !== projectId || entry.agentId !== agentId) continue; + entry.approvals.denyAll(); + entry.abort?.abort(); + if (entry.running) runnings.push(entry.running); + this.entries.delete(key); + } + return runnings; + } + + endAgentDeletion(projectId: string, agentId: string): void { + this.deletingAgents.delete(agentKey(projectId, agentId)); + } + + /** + * Before deleting a single Session, converge its active run and clear it out of the + * active table (same semantics as beginAgentDeletion). Also marks this Session as + * "being deleted": new Tasks/compactions entering during the deletion process are + * always rejected with 409 (assertSessionNotDeleting), closing the race window where + * a new task recreates the entry and Trace file, reviving an already-deleted Session + * between the abort snapshot and the file removal. The caller must call + * endSessionDeletion once deletion finishes (success or failure). Returns the + * in-flight drive Promise: the caller should await it before deleting the Trace file, + * so cleanup writes don't recreate the file. + */ + beginSessionDeletion(sessionId: string): Promise[] { + this.deletingSessions.add(sessionId); + const entry = this.entries.get(sessionId); + if (!entry) return []; + entry.approvals.denyAll(); + entry.abort?.abort(); + this.entries.delete(sessionId); + return entry.running ? [entry.running] : []; + } + + endSessionDeletion(sessionId: string): void { + this.deletingSessions.delete(sessionId); + } + + /** Graceful shutdown: reject new tasks (503), interrupt all active runs, and wait for them to finish (default ≤5s). */ + async shutdown(timeoutMs = 5000): Promise { + this.closed = true; + clearInterval(this.sweepTimer); + const pending: Promise[] = []; + for (const entry of this.entries.values()) { + if (!entry.abort) continue; + entry.approvals.denyAll(); + entry.abort.abort(); + if (entry.running) pending.push(entry.running); + } + if (pending.length === 0) return; + await Promise.race([ + Promise.allSettled(pending).then(() => undefined), + new Promise((resolve) => setTimeout(resolve, timeoutMs).unref?.()), + ]); + } + + /** + * Active-table idle eviction: removes entries that are idle (idle status, no pending + * approvals, no in-flight drive) and have been inactive past the timeout, releasing + * the core Session's full in-memory history. This is purely memory reclamation: the + * next access re-resumes via the loader, so correctness is unaffected. Lock-table + * entries are auto-cleaned by withLock once their chain drains (including leftover + * entries under the old id after self-heal). `now` / `idleMs` are injectable for + * tests and timers. + */ + sweepIdle(now: number = Date.now(), idleMs: number = ENTRY_IDLE_MS): void { + for (const [key, entry] of this.entries) { + if (entry.status !== "idle" || entry.approvals.size !== 0 || entry.running !== null) continue; + if (now - entry.lastActivityMs <= idleMs) continue; + this.entries.delete(key); + } + } + + // —— Internal —— + + private assertOpen(): void { + if (this.closed) { + throw new HttpError(503, "shutting_down", "服务端正在关停,暂不接受新任务。"); + } + } + + /** The Agent owning this Session is being deleted → 409 (guards against directory recreation inside the deletion race window). */ + private assertAgentNotDeleting(sessionId: string): void { + const row = this.deps.sessions.findById(sessionId); + if (row && this.deletingAgents.has(agentKey(row.projectId, row.agentId))) { + throw new HttpError(409, "agent_deleting", "该 Agent 正在删除,暂不接受新任务。"); + } + } + + /** This Session is being deleted → 409 (guards against the entry/Trace being rebuilt and reviving it inside the deletion race window). */ + private assertSessionNotDeleting(sessionId: string): void { + if (this.deletingSessions.has(sessionId)) { + throw new HttpError(409, "session_deleting", "该 Session 正在删除,暂不接受新任务。"); + } + } + + private assertIdle(entry: RuntimeEntry): void { + if (entry.status === "running") { + throw new HttpError(409, "task_in_progress", "该 Session 已有进行中的 Task。"); + } + if (entry.status === "compacting") { + throw new HttpError(409, "compacting", "该 Session 正在压缩上下文,暂不接受新输入。"); + } + } + + /** get-or-resume-or-heal: use directly on an active-table hit; otherwise load via the loader, updating the index's primary key on self-heal. */ + private async ensureEntry(sessionId: string): Promise { + const existing = this.entries.get(sessionId); + if (existing) return existing; + const row = this.deps.sessions.findById(sessionId); + if (!row) { + throw new HttpError(404, "session_not_found", "Session 不存在或无权访问。"); + } + const session = await this.deps.loader.load(row); + // The Session/Agent was marked for deletion while loading: discard the load result, + // don't rebuild the entry (avoids reviving an orphaned Trace). + this.assertSessionNotDeleting(row.sessionId); + this.assertAgentNotDeleting(row.sessionId); + let currentId = row.sessionId; + if (session.sessionId !== row.sessionId) { + // Self-heal produced a new session_id: update the index's primary key; the SSE + // channel and pending state are naturally empty for it. + this.deps.sessions.replaceId(row.sessionId, session.sessionId); + currentId = session.sessionId; + } + const entry: RuntimeEntry = { + sessionId: currentId, + projectId: row.projectId, + agentId: row.agentId, + provider: row.provider, + modelId: row.modelId, + session, + status: "idle", + approvals: new ApprovalRegistry(), + abort: null, + running: null, + lastActivityMs: Date.now(), + }; + this.entries.set(currentId, entry); + return entry; + } + + /** + * Drive the output stream in the background: publish each message + persist usage + + * persist LLM/tool errors; on completion (including errors) resets to idle and pushes + * the status. `titleSource` is passed only for Task runs (compaction doesn't generate + * a title): it collects model text for automatic title generation. + */ + private async drive( + entry: RuntimeEntry, + gen: AsyncGenerator, + titleSource?: { userExcerpt: string }, + ): Promise { + const ctx: UsageContext = { + projectId: entry.projectId, + agentId: entry.agentId, + sessionId: entry.sessionId, + provider: entry.provider, + modelId: entry.modelId, + }; + // LLM request failures and tool execution failures aren't expressed via throw (core + // converges them into the message stream), so the try/catch below can't catch them: + // the watcher inspects messages one by one and fishes them out for persistence + // (subagent failures flow through this same stream too; see stream-error-watcher). + const watcher = this.deps.errors + ? new StreamErrorWatcher(this.deps.errors, { + projectId: entry.projectId, + agentId: entry.agentId, + sessionId: entry.sessionId, + }) + : null; + // Subagent (origin) registration: as soon as session_meta arrives, the child Session + // is persisted so it appears immediately in the sidebar (the frontend picks it up + // when it refreshes the list at task completion). The title material is "the prompt + // of the run_subagent call that spawned this subagent" — the subagent's user input + // is never replayed onto the parent stream (ContextEngine writes the Trace but never + // yields it), so we can't rely on the subagent's first user message; instead we use + // the run_subagent tool_call arguments immediately preceding it on the parent stream + // (depth limited to 1, spawned in order, so taking the most recent one suffices). + /** Subagents registered during this run (keyed by session id); titles are generated for each on completion. */ + const children = new Map(); + // Unclaimed run_subagent prompts, queued in call order: a single round may spawn + // multiple subagents in parallel, and a subagent's session_meta only carries the + // session id (no tool_call_id), so pairing can only be approximated via FIFO (when + // spawned in parallel and session_meta arrives out of order, two subagents' titles + // may end up swapped — this only affects the displayed title). A call that will + // never produce a subagent must be dequeued, or its prompt would be mismatched onto + // the next subagent: this covers denied calls (approval_decision ≠ allow), and calls + // that were approved but failed before spawning the subagent (e.g. agent_id doesn't + // exist) — the latter is cleaned up when the parent-level tool_call_output settles; + // if the call is still in the queue at that point, it never produced a session_meta. + const subagentPrompts = new Map(); + try { + for await (const msg of gen) { + // A parent-level (no origin) run_subagent call: record its prompt for the child + // session_meta that arrives later to use as its title. + if (!msg.origin || msg.origin.length === 0) { + const call = runSubagentCall(msg); + if (call) subagentPrompts.set(call.toolCallId, call.prompt); + const denied = deniedToolCallId(msg); + if (denied) subagentPrompts.delete(denied); + const settled = settledToolCallId(msg); + if (settled) subagentPrompts.delete(settled); + } else if (isSessionMeta(msg)) { + // Subagent registration is only a "side effect" — it must never interrupt the + // main run flow on error: wrap the whole thing in a defensive try/catch. + try { + const child = this.registerChildSession(entry, msg, children); + // Only a **direct** subagent (origin length 1) claims a queued parent-level + // run_subagent prompt; deeper sessions are spawned by their own parent and + // shouldn't consume from this queue. + if (child && msg.origin!.length === 1) { + const [pendingId] = subagentPrompts.keys(); + if (pendingId !== undefined) { + child.prompt = subagentPrompts.get(pendingId) ?? ""; + subagentPrompts.delete(pendingId); // Consumed by this session_meta + } + } + } catch (err) { + this.log( + `[subagent] 子会话登记失败: ${err instanceof Error ? err.message : String(err)}`, + ); + this.deps.errors?.record({ + source: "subagent", + err, + ctx, + code: "subagent_register_failed", + }); + } + } else { + // A subagent's model text: its title is generated from the subagent's **own + // conversation**, so the material is accumulated here. + const nested = nestedAssistantText(msg); + const child = nested ? children.get(nested.sessionId) : undefined; + if (nested && child && child.assistantExcerpt.length < TITLE_EXCERPT_LIMIT) { + child.assistantExcerpt += (child.assistantExcerpt ? "\n" : "") + nested.text; + } + } + // Re-fetch the channel before every publish (matches publishEvent): the channel + // may have been recycled and recreated during a long wait on approval, and + // holding a stale reference would send output to an orphaned, detached channel. + this.deps.channels.get(entry.sessionId).publish(msg); + watcher?.observe(msg); + try { + await this.deps.recorder.record(ctx, msg); + } catch (err) { + this.log(`[usage] 落库失败: ${err instanceof Error ? err.message : String(err)}`); + this.deps.errors?.record({ source: "usage", err, ctx, code: "usage_insert_failed" }); + } + } + } catch (err) { + // The SDK doesn't normally throw (errors are converged into the message stream); + // this is a defensive record here to avoid crashing the runtime. + this.log( + `[session] 运行异常: ${err instanceof Error ? (err.stack ?? err.message) : String(err)}`, + ); + this.deps.errors?.record({ source: "session", err, ctx, code: "session_run_failed" }); + } finally { + // Wrap-up: persist any still-pending LLM failure and clear the tool-name cache (the watcher's state doesn't carry across runs). + watcher?.close(); + entry.approvals.denyAll(); + entry.status = "idle"; + entry.abort = null; + entry.running = null; + entry.lastActivityMs = Date.now(); + this.publishState(entry, "idle"); + if (titleSource && titleSource.userExcerpt.trim()) { + // Attempt generation whenever there's user material; whether generation is + // actually needed (title still NULL, etc.) is decided by the generator itself. + // Material is collected by the core Session during run; here we only pass the + // fallback text. + this.deps.titles?.maybeGenerate(ctx, entry.session, { + fallbackText: titleSource.userExcerpt, + }); + } + // A subagent's title is likewise generated by the model, with material being the + // subagent's **own conversation**: the prompt that spawned it plus its own reply + // (material the parent Session collects belongs to the parent, hence the explicit + // override here). It piggybacks a one-shot request on the parent Session's bare + // LLM (the child Session object never leaves the SDK); on failure/empty result the + // generator falls back to the prompt's first line. + for (const child of children.values()) { + if (!child.prompt.trim()) continue; + this.deps.titles?.maybeGenerate( + // Bookkeeping: Session/Agent record the subagent (the title belongs to it), + // but the model reference still uses ctx's **parent-Session** pair + // (provider, modelId) — this one-shot request really does run on the parent + // Session's bare LLM (a subagent may switch models via run_subagent's + // model_id). + { ...ctx, agentId: child.agentId, sessionId: child.sessionId }, + entry.session, + { + fallbackText: child.prompt, + material: { userText: child.prompt, assistantText: child.assistantExcerpt }, + notifyOn: entry.sessionId, // Notify the frontend via the parent Session's SSE channel + }, + ); + } + } + } + + /** + * Register a subagent: persisted only when the origin message is session_meta + * (agentId is derived from the agent_state path: `<…>//agent_state`). + * **The title is left blank** — it's generated at the end of this run by the model + * from the subagent's own conversation (see drive's finally), falling back to the + * first line of the run_subagent prompt if generation fails. Idempotent (children + * dedup + insertOrIgnore); a subagent has its own Trace, so it's visible in both the + * list and the trace view. On successful registration, the entry is put into + * `children` and returned; a duplicate session_meta returns null. + */ + private registerChildSession( + entry: RuntimeEntry, + msg: OmniMessage, + children: Map, + ): ChildSession | null { + if (!isSessionMeta(msg)) return null; + const childSid = msg.origin![msg.origin!.length - 1]!; + if (children.has(childSid)) return null; + const p = msg.payload as SessionMetaPayload; + const agentId = path.basename(path.dirname(p.agent_state)); + if (!agentId || agentId === "." || agentId === "..") return null; + this.deps.sessions.insertOrIgnore({ + sessionId: childSid, + projectId: entry.projectId, + agentId, + provider: p.provider, + modelId: p.model_id, + workspace: p.workspace, + // A subagent's approvals are inherited from the parent Session; the index row is + // inserted with defaults (matches the convention for Sessions discovered by the CLI). + approvalMode: "allow-all", + title: null, + source: "subagent", + createdAt: new Date().toISOString(), + }); + // Make the subagent appear immediately in the sidebar: notify via the parent + // Session's channel (a frontend currently watching the parent run refreshes its list in place). + this.publishEvent(entry, { + type: "session_created", + projectId: entry.projectId, + agentId, + sessionId: childSid, + source: "subagent", + }); + const child: ChildSession = { + sessionId: childSid, + agentId, + modelId: p.model_id, + prompt: "", + assistantExcerpt: "", + }; + children.set(childSid, child); + return child; + } + + private publishState(entry: RuntimeEntry, state: SessionStatus): void { + this.publishEvent(entry, { type: "task_state", state }); + } + + private publishEvent(entry: RuntimeEntry, event: ServerEvent): void { + this.deps.channels.get(entry.sessionId).publish(event, "server_event"); + } + + /** Serialize (mutually exclude) execution by sessionId; cleans up the lock-table entry once its chain drains (avoids unbounded growth). */ + private async withLock(sessionId: string, fn: () => Promise): Promise { + // What's stored in the chain is the already-caught version (used only for + // sequencing, never propagates errors); the caller gets the original result from `next`. + const prev = this.locks.get(sessionId) ?? Promise.resolve(); + const next = prev.then(fn); + const settled: Promise = next + .then( + () => undefined, + () => undefined, + ) + .then(() => { + // Only delete if still the tail of the chain (no later waiter): preserves mutual-exclusion semantics. + if (this.locks.get(sessionId) === settled) this.locks.delete(sessionId); + }); + this.locks.set(sessionId, settled); + return next; + } +} diff --git a/packages/server/src/runtime/stream-error-watcher.ts b/packages/server/src/runtime/stream-error-watcher.ts new file mode 100644 index 0000000..1fcc7b2 --- /dev/null +++ b/packages/server/src/runtime/stream-error-watcher.ts @@ -0,0 +1,274 @@ +/** + * Error capture within the message stream: LLM request + * failures and tool execution failures are **both never expressed via throw** — core + * converges them into the message stream (LLM and Environment handle errors + * internally and never throw), so a try/catch can't catch a single one. This watcher + * hooks onto SessionManager's drive, inspects messages one by one, and fishes them out + * into error_records (source = `llm` / `environment`), matching usage-recorder's shape: + * recognizes only a few payload types, no-op on the rest. **One instance per run/compact** + * (its state wraps up accordingly, see close). + * + * LLM (source = `llm`): reads the status of `request_end` — + * - `failed` → unexpected (not retryable: auth failure, invalid params, etc., needs a human); + * - `timeout` / `malformed` → expected (the engine already reconnects and retries, part + * of normal operation); + * - `aborted` / `completed` are not recorded (the former is a user-initiated interrupt, + * not an error). + * + * The message uses the real reason: `request_end` only carries status, and **the only + * place core carries the actual failure-reason text is the `abort` event's reason** + * (e.g. `llm request error: 401 …` / `malformed response failed after N retries`). So a + * `request_end` failure is first held pending, not persisted immediately, and is + * resolved at the next request boundary: + * - Immediately followed by `abort` → use its reason as the message (the real reason); + * - Immediately followed by `request_begin` (the engine is retrying) → no reason text + * left to wait for, use the status text; + * - Still unresolved when the run ends → close persists it as a fallback. + * Exception: when reason is a user-interrupt message (`aborted …`), it's not trusted — + * "the user clicked stop during backoff" isn't the reason for this timeout, so the + * status text is used instead (that timeout is a genuine failure and is still recorded). + * Pending state is bucketed by origin: subagent messages interleave with the parent + * session's (even more so with parallel subagents), and mixing them up would misattribute. + * + * Environment (source = `environment`): reads `tool_call_output`'s stop_reason ∈ + * {failed, timeout} → expected (the error is fed back to the model, and the Agent + * adjusts on its own; `aborted` is denial/interruption, not recorded). + * `tool_call_output` only has tool_call_id, no tool name, so `tool_call_id → tool name` + * is cached (tool_call always arrives before its output), and the tool name is written + * into code (`tool_failed:exec_command`) — so the stats dashboard's "most common error + * code" and the error table can show at a glance which tool failed. + * + * **Attribution (ctx) is recorded against the session that actually produced the error, + * not always the parent Session**: a subagent's LLM failures and tool failures also flow + * through this same stream (carrying origin); if we simply reused the parent ctx passed + * in at construction, filtering errors by Agent would always show 0 for the child Agent + * and an inflated count for the parent — both attribution stats and the troubleshooting + * target would be wrong. So we recognize `session_meta` carrying origin (a subagent's + * first message, always arriving before any of its failures), registering + * `origin → {agentId, sessionId}` (agentId derived from the agent_state path, matching + * SessionManager.registerChildSession's convention); at persist time we look up the + * message's origin: a hit records the subagent, a miss (a main-session message, or + * session_meta hasn't arrived yet) falls back to the parent ctx. projectId is always + * taken from the parent — a subagent is necessarily in the same Project. + */ +import { isEventMessage, isModelMessage, isSessionMeta } from "@prismshadow/penguin-core"; +import path from "node:path"; +import type { OmniMessage, SessionMetaMessage, StopReason } from "@prismshadow/penguin-core"; +import { MESSAGE_MAX } from "./error-recorder.js"; +import type { ErrorContext, ErrorKind, ErrorSink } from "./error-recorder.js"; + +/** Cap on the tool-name cache (bounded, to prevent unbounded growth over a long run; over the limit, evicts the oldest by registration order). */ +export const TOOL_NAMES_MAX = 1000; + +/** + * Cap on the subagent-identity cache (bounded, same reasoning as TOOL_NAMES_MAX: prevent + * unbounded growth over a long run). The number of in-flight subagents is naturally + * bounded by the subagent concurrency limit and falls far short of this value; over the + * limit, evicts the oldest by registration order (those subagents have long since + * settled, so even if a failure still arrives, it just falls back to the parent ctx — + * i.e., the pre-fix behavior). + */ +export const ORIGIN_CTX_MAX = 200; + +/** Recorded LLM failure states (`aborted` / `completed` are not errors and aren't included here). */ +type LlmFailure = "failed" | "timeout" | "malformed"; + +/** LLM failure state → error code, classification, and fallback message (used when the abort reason isn't available). */ +const LLM_FAILURES: Record = { + failed: { code: "llm_failed", kind: "unexpected", text: "LLM 请求失败(不可重试)。" }, + timeout: { code: "llm_timeout", kind: "expected", text: "LLM 请求超时(引擎重连重试)。" }, + malformed: { + code: "llm_malformed", + kind: "expected", + text: "LLM 响应无法解析(引擎重连重试)。", + }, +}; + +/** Recorded tool failure states (`aborted` = denial/interruption, not an error). */ +type ToolFailure = "failed" | "timeout"; + +function isLlmFailure(s: unknown): s is LlmFailure { + return s === "failed" || s === "timeout" || s === "malformed"; +} + +function isToolFailure(s: unknown): s is ToolFailure { + return s === "failed" || s === "timeout"; +} + +/** A user-interrupt abort message (core's `aborted by user` / `aborted during …`): not a failure reason. */ +function isUserAbortReason(reason: string): boolean { + return /^aborted\b/i.test(reason); +} + +/** The session a message belongs to (last origin element; empty string for the main session) — both pending state and the tool-name cache are bucketed by it. */ +function originKey(msg: OmniMessage): string { + const origin = msg.origin; + return origin && origin.length > 0 ? origin[origin.length - 1]! : ""; +} + +/** + * Take the **tail** of the tool output (not the head) as the message: core appends the + * failure reason (`[tool error] …` / `[tool timeout: …]` / exit code) at the end of the + * output, so truncating from the head would leave only a chunk of normal stdout and + * drop the reason. + */ +function toolFailureText(output: string): string { + if (!output) return "工具执行失败(无输出)。"; + if (output.length <= MESSAGE_MAX) return output; + return `…${output.slice(output.length - (MESSAGE_MAX - 1))}`; +} + +export class StreamErrorWatcher { + /** LLM failures awaiting a real reason: origin → failure state (see file header; each session has at most one in-flight Request). */ + private readonly pending = new Map(); + /** Names of in-flight tool calls: `origin \0 tool_call_id` → tool name (dequeued once output arrives). */ + private readonly toolNames = new Map(); + /** Subagent identity: origin → that subagent's `{agentId, sessionId}` (see the file header's attribution section). */ + private readonly originCtx = new Map(); + + constructor( + private readonly errors: ErrorSink, + private readonly ctx: ErrorContext, + ) {} + + /** Consume one outgoing message; messages irrelevant to this watcher are a no-op. */ + observe(msg: OmniMessage): void { + if (isSessionMeta(msg)) { + this.registerOrigin(msg); + return; + } + if (isModelMessage(msg)) { + this.observeTool(msg); + return; + } + if (isEventMessage(msg)) this.observeLlm(msg); + } + + /** run/compact wrap-up: persist any still-pending failure (that never got its abort), and clear caches (prevents leaks). */ + close(): void { + for (const key of [...this.pending.keys()]) this.flush(key); + this.pending.clear(); + this.toolNames.clear(); + this.originCtx.clear(); + } + + // —— Attribution (see file header) —— + + /** + * Register a subagent's identity: `session_meta` carrying origin is the subagent's + * first message, always arriving before any of its failures. agentId is derived from + * the absolute agent_state path (`<…>//agent_state`) — matching + * SessionManager.registerChildSession's convention; not registered if the path is + * malformed (that subagent's failures fall back to the parent ctx — better to + * misattribute than write into a nonexistent agentId). The main session's session_meta + * (no origin) is already the parent ctx and isn't registered. + */ + private registerOrigin(msg: SessionMetaMessage): void { + const key = originKey(msg); + if (!key) return; // Main session + const agentId = path.basename(path.dirname(msg.payload.agent_state)); + if (!agentId || agentId === "." || agentId === "..") return; + // Bounded (re-registering refreshes registration order; over the limit, evicts the oldest). + this.originCtx.delete(key); + this.originCtx.set(key, { agentId, sessionId: key }); + if (this.originCtx.size > ORIGIN_CTX_MAX) { + const oldest = this.originCtx.keys().next().value; + if (oldest !== undefined) this.originCtx.delete(oldest); + } + } + + /** + * The attribution to persist for this origin: a registered subagent hit → record its + * own Agent/Session; a miss (a main-session message, or session_meta hasn't arrived + * yet) → fall back to the parent ctx passed at construction. projectId is always taken + * from the parent (a subagent is necessarily in the same Project). + */ + private ctxFor(key: string): ErrorContext { + const child = this.originCtx.get(key); + if (!child) return this.ctx; + return { projectId: this.ctx.projectId, agentId: child.agentId, sessionId: child.sessionId }; + } + + // —— LLM —— + + private observeLlm(msg: OmniMessage): void { + const p = msg.payload as { type?: string; status?: StopReason; reason?: string | null }; + const key = originKey(msg); + if (p.type === "request_end") { + this.flush(key); // Defensive: if a previous failure is still pending (normally resolved by request_begin), persist it first + if (isLlmFailure(p.status)) this.pending.set(key, p.status); + return; + } + // A new attempt begins (the engine is retrying): no reason text left to wait for the previous failure, persist using the status text. + if (p.type === "request_begin") { + this.flush(key); + return; + } + // Interrupted/failed exit: reason is core's only failure-reason text. + if (p.type === "abort") { + this.flush(key, typeof p.reason === "string" ? p.reason : null); + } + } + + /** + * Persist a pending LLM failure (no-op if none is pending); `reason` is the abort + * message that arrived afterward. Pending state is already bucketed by origin, so + * `key` is exactly "the session that produced this failure" — attribution is looked + * up from it (see file header). + */ + private flush(key: string, reason?: string | null): void { + const status = this.pending.get(key); + if (status === undefined) return; + this.pending.delete(key); + const spec = LLM_FAILURES[status]; + const trimmed = reason?.trim(); + // A user-interrupt message isn't a failure reason (see file header); fall back to the status text. + const message = trimmed && !isUserAbortReason(trimmed) ? trimmed : spec.text; + this.errors.record({ + source: "llm", + err: message, + ctx: this.ctxFor(key), + code: spec.code, + kind: spec.kind, + }); + } + + // —— Environment (tool execution) —— + + private observeTool(msg: OmniMessage): void { + const p = msg.payload as { + type?: string; + name?: string; + output?: string; + tool_call_id?: string; + stop_reason?: StopReason; + }; + if (typeof p.tool_call_id !== "string") return; + const origin = originKey(msg); // The session that made this call (both attribution and the tool-name cache are bucketed by it) + const key = `${origin}\0${p.tool_call_id}`; + + if (p.type === "tool_call" && typeof p.name === "string") { + // tool_call arrives before its output: record the tool name (bounded, re-registering refreshes registration order). + this.toolNames.delete(key); + this.toolNames.set(key, p.name); + if (this.toolNames.size > TOOL_NAMES_MAX) { + const oldest = this.toolNames.keys().next().value; + if (oldest !== undefined) this.toolNames.delete(oldest); + } + return; + } + if (p.type !== "tool_call_output") return; + + const name = this.toolNames.get(key); + this.toolNames.delete(key); // This call has settled: dequeue it, the cache only keeps in-flight calls + if (!isToolFailure(p.stop_reason)) return; // completed / aborted (denial, user interrupt) are not errors + this.errors.record({ + source: "environment", + err: toolFailureText(p.output ?? ""), + ctx: this.ctxFor(origin), + // The tool name goes into code: so the stats dashboard's "most common error code" and table can show which tool failed. + code: `tool_${p.stop_reason}:${name ?? "unknown"}`, + kind: "expected", + }); + } +} diff --git a/packages/server/src/runtime/title-generator.ts b/packages/server/src/runtime/title-generator.ts new file mode 100644 index 0000000..601ccd8 --- /dev/null +++ b/packages/server/src/runtime/title-generator.ts @@ -0,0 +1,137 @@ +/** + * The **policy layer** for automatic Session title generation (conversation-page + * extension). + * + * "How to generate" lives in the core SDK (`session.generateTitle`: an out-of-band + * one-shot request on the session's own Model, no tools, thinking disabled, writes no + * history/Trace); this module is only responsible for host-side policy: + * - When to generate: after a Task completes and the DB row's title is still NULL + * (i.e. after the first successful conversation); + * - Persistence and notification: writes sessions.title and pushes a `session_title` + * server event to the Session channel; + * - Bookkeeping: the one-shot request's token consumption is converted to token_usage + * and handed to usage-recorder for persistence; + * - Silent failure (logged): the title stays NULL and naturally retries after the next + * Task completes. + */ +import { emptyTokenCounts, sanitizeTitle, tokenUsage } from "@prismshadow/penguin-core"; +import type { SessionsRepo } from "../db/repos/sessions.js"; +import type { ChannelHub } from "./channel.js"; +import type { ErrorSink } from "./error-recorder.js"; +import type { RuntimeSession } from "./session-manager.js"; +import type { UsageContext, UsageRecorder } from "./usage-recorder.js"; + +export interface TitleGeneratorDeps { + sessions: SessionsRepo; + channels: ChannelHub; + recorder: Pick; + /** Error persistence (optional: without it, only logs — same as before this was wired up). */ + errors?: ErrorSink; + log?: (line: string) => void; +} + +/** Host-side parameters for one title-generation request. */ +export interface TitleRequest { + /** Fallback material for when the LLM fails or returns an empty result (cleaned and truncated from the first non-empty line). */ + fallbackText: string; + /** Material override (for subagents — the material is the subagent's own conversation); defaults to the first Task's material self-collected by the core Session. */ + material?: { userText: string; assistantText: string }; + /** The channel to push the `session_title` event to; defaults to `ctx.sessionId`. A + * subagent has no SSE channel of its own, so its title must reach the frontend via + * the **parent Session's** channel (the list updates in place by sessionId). */ + notifyOn?: string; +} + +/** session-manager's minimal dependency on the title generator (tests inject a fake implementation). */ +export interface TitleNotifier { + maybeGenerate( + ctx: UsageContext, + session: Pick, + req: TitleRequest, + ): void; +} + +export class TitleGenerator implements TitleNotifier { + private readonly inflight = new Set(); + private readonly log: (line: string) => void; + + constructor(private readonly deps: TitleGeneratorDeps) { + this.log = deps.log ?? ((line) => console.error(line)); + } + + /** Generate a title in the background (fire-and-forget) when conditions are met: the row exists, title is still NULL, and no generation is already in flight. */ + maybeGenerate( + ctx: UsageContext, + session: Pick, + req: TitleRequest, + ): void { + const row = this.deps.sessions.findById(ctx.sessionId); + if (!row || row.title !== null) return; + if (this.inflight.has(ctx.sessionId)) return; + this.inflight.add(ctx.sessionId); + void this.generate(ctx, session, req) + .catch((err: unknown) => { + this.log(`[title] 生成失败: ${err instanceof Error ? err.message : String(err)}`); + this.deps.errors?.record({ source: "title", err, ctx, code: "title_failed" }); + }) + .finally(() => { + this.inflight.delete(ctx.sessionId); + }); + } + + private async generate( + ctx: UsageContext, + session: Pick, + req: TitleRequest, + ): Promise { + let title: string | null = null; + try { + // Material defaults to what the core Session self-collects during run; it's only + // overridden in scenarios like subagents where the material isn't on that Session. + const res = await session.generateTitle( + req.material ? { material: req.material } : undefined, + ); + title = res.title; + // The one-shot request's real consumption is metered as usual (converted to token_usage and handed to recorder, attributed to this Session). + if (res.usage) { + try { + await this.deps.recorder.record(ctx, tokenUsage(emptyTokenCounts(), res.usage)); + } catch (err) { + this.log(`[title] 用量落库失败: ${err instanceof Error ? err.message : String(err)}`); + this.deps.errors?.record({ + source: "title", + err, + ctx, + code: "title_usage_insert_failed", + }); + } + } + } catch (err) { + // A model request error (rate limit / timeout / network, etc.) shouldn't leave the + // title permanently missing: log it and fall through to the fallback. + this.log(`[title] 模型请求失败: ${err instanceof Error ? err.message : String(err)}`); + this.deps.errors?.record({ source: "title", err, ctx, code: "title_llm_failed" }); + } + // When the LLM produces no usable title (failure / empty result), truncate the fallback material's first line — this guarantees a title is always generated. + const finalTitle = title ?? fallbackTitle(req.fallbackText); + if (finalTitle === null) return; + // There may already be a concurrent write during generation (e.g. a future manual rename): only persist if still NULL. + const latest = this.deps.sessions.findById(ctx.sessionId); + if (!latest || latest.title !== null) return; + this.deps.sessions.updateTitle(ctx.sessionId, finalTitle); + this.deps.channels + .get(req.notifyOn ?? ctx.sessionId) + .publish( + { type: "session_title", sessionId: ctx.sessionId, title: finalTitle }, + "server_event", + ); + } +} + +/** Fallback title: take the material's first non-empty line, sanitize and truncate; if sanitizing empties it out (pure punctuation, etc.) fall back to the truncated original text; returns null if all-whitespace. */ +function fallbackTitle(text: string): string | null { + const firstLine = text.split("\n").find((l) => l.trim().length > 0); + if (!firstLine) return null; + // sanitizeTitle strips a pure-punctuation line down to empty — in that case keep the truncated original text, guaranteeing "a title is always obtained". + return sanitizeTitle(firstLine) ?? firstLine.trim().slice(0, 30); +} diff --git a/packages/server/src/runtime/usage-recorder.ts b/packages/server/src/runtime/usage-recorder.ts new file mode 100644 index 0000000..e2e0345 --- /dev/null +++ b/packages/server/src/runtime/usage-recorder.ts @@ -0,0 +1,116 @@ +/** + * Usage persistence. + * + * Consumes the Session output stream: + * - A subagent's `session_meta` (carrying origin) → registers the mapping "origin's + * last session_id → (provider, model_id)" (a subagent may use a different Model, so + * cost is priced against the actual Model used); + * - `token_usage` → inserts one usage_records row: the four token fields come from + * `payload.request` (per-Request increment; subagents are recorded one row at a + * time, with `session_id` attributed to their owning main Session). Only tokens are + * persisted, not cost — pricing may be added later, and cost is computed on the fly + * against current pricing when usage-service queries. + * The attribution key is always the paired reference `(provider, model_id)` (the same + * model_id name across different vendors is attributed separately). + */ +import { isEventMessage, isSessionMeta } from "@prismshadow/penguin-core"; +import type { OmniMessage } from "@prismshadow/penguin-core"; +import { formatLocalDate } from "../internal/dates.js"; +import type { UsageRepo } from "../db/repos/usage.js"; + +/** Attribution context for one record (top-level Session scope). */ +export interface UsageContext { + projectId: string; + agentId: string; + /** Top-level Session id (the current actual id after self-heal). */ + sessionId: string; + /** Vendor grouping for the top-level Session's model (paired with modelId; the fallback attribution when the origin mapping has no hit). */ + provider: string; + /** Upstream model_id of the top-level Session (paired with provider). */ + modelId: string; +} + +/** Cap on the subagent attribution mapping: over the limit, evicts the oldest by insertion order (an evicted entry falls back to the main Session's Model attribution). */ +export const ORIGIN_MODELS_MAX = 1000; + +export class UsageRecorder { + /** Subagent model attribution mapping: origin's last session_id → paired reference (session_id is globally unique). */ + private readonly originModels = new Map(); + + constructor( + private readonly usage: UsageRepo, + private readonly now: () => Date = () => new Date(), + ) {} + + /** Consume one outgoing message; messages other than session_meta / token_usage are a no-op. */ + async record(ctx: UsageContext, msg: OmniMessage): Promise { + if (isSessionMeta(msg) && msg.origin && msg.origin.length > 0) { + const originSessionId = msg.origin[msg.origin.length - 1]!; + // Bounded mapping (avoids unbounded growth over a long-running process): re-inserting refreshes insertion order, over the limit evicts the oldest. + this.originModels.delete(originSessionId); + this.originModels.set(originSessionId, { + provider: msg.payload.provider, + modelId: msg.payload.model_id, + }); + if (this.originModels.size > ORIGIN_MODELS_MAX) { + const oldest = this.originModels.keys().next().value; + if (oldest !== undefined) this.originModels.delete(oldest); + } + return; + } + if (!isEventMessage(msg)) return; + const payload = msg.payload as { + type?: string; + request?: { cache_read: number; cache_write: number; output: number; total: number }; + status?: string; + }; + + const originSessionId = + msg.origin && msg.origin.length > 0 ? msg.origin[msg.origin.length - 1]! : null; + // Empty origin → main session's Model; otherwise look up the mapping, falling back to the main Session's Model (paired) on a miss. + const ref = + originSessionId === null + ? { provider: ctx.provider, modelId: ctx.modelId } + : (this.originModels.get(originSessionId) ?? { + provider: ctx.provider, + modelId: ctx.modelId, + }); + const now = this.now(); + const base = { + ts: now.toISOString(), + date: formatLocalDate(now), + projectId: ctx.projectId, + agentId: ctx.agentId, + sessionId: ctx.sessionId, + originSessionId, + provider: ref.provider, + modelId: ref.modelId, + }; + + if (payload.type === "token_usage" && payload.request) { + // A successful request: persist along with tokens (status defaults to completed). + const r = payload.request; + this.usage.insert({ + ...base, + cacheRead: r.cache_read, + cacheWrite: r.cache_write, + output: r.output, + total: r.total, + }); + return; + } + // A failed request (request_end and not completed, usually with no token_usage): + // persist 0 tokens + status, feeding the "model success rate" stat; a successful + // request is already counted once via the token_usage branch above, not repeated here. + if (payload.type === "request_end" && payload.status && payload.status !== "completed") { + this.usage.insert({ + ...base, + cacheRead: 0, + cacheWrite: 0, + output: 0, + total: 0, + status: payload.status, + }); + } + } +} diff --git a/packages/server/src/services/admin-service.ts b/packages/server/src/services/admin-service.ts new file mode 100644 index 0000000..c9f3de2 --- /dev/null +++ b/packages/server/src/services/admin-service.ts @@ -0,0 +1,96 @@ +/** + * Admin user backend: user list / create / reset password / delete. + * + * - Create: username is the user_id (^[a-z][a-z0-9_-]{1,31}$); admin sets the initial + * password, flagged with password_is_initial; a default Project `proj-` is + * auto-created, rolling back the user row on failure. + * - Reset password: also flags the password as initial and clears all of the user's + * login sessions (forcing re-login). + * - Delete: the built-in admin cannot be deleted; Projects owned by the user are + * deleted along with it (including data directories), with sessions/memberships/UI + * preferences cascade-deleted via foreign keys. + */ +import type { UserInfo } from "../api/types.js"; +import { HttpError } from "../http/errors.js"; +import { MIN_PASSWORD_LENGTH, toUserInfo } from "../auth/service.js"; +import { hashPassword } from "../auth/password.js"; +import type { AuthSessionsRepo } from "../db/repos/auth-sessions.js"; +import type { ProjectsRepo } from "../db/repos/projects.js"; +import type { UserRow, UsersRepo } from "../db/repos/users.js"; +import { SEMANTIC_ID_RULE, USERNAME_PATTERN } from "./ids.js"; +import type { ProjectService } from "./project-service.js"; + +export interface AdminServiceDeps { + users: UsersRepo; + authSessions: AuthSessionsRepo; + projects: ProjectsRepo; + projectService: ProjectService; + now?: () => Date; +} + +export class AdminService { + private readonly now: () => Date; + + constructor(private readonly deps: AdminServiceDeps) { + this.now = deps.now ?? (() => new Date()); + } + + listUsers(): UserInfo[] { + return this.deps.users.list().map(toUserInfo); + } + + async createUser(userId: string, password: string): Promise { + if (!USERNAME_PATTERN.test(userId)) { + throw new HttpError(400, "invalid_user_id", `用户名须为 2~32 位:${SEMANTIC_ID_RULE}。`); + } + if (password.length < MIN_PASSWORD_LENGTH) { + throw new HttpError(400, "invalid_password", "密码至少 8 个字符。"); + } + if (this.deps.users.findById(userId)) { + throw new HttpError(409, "user_exists", `用户已存在:${userId}。`); + } + const user: UserRow = { + userId, + passwordHash: await hashPassword(password), + isAdmin: false, + passwordIsInitial: true, + createdAt: this.now().toISOString(), + }; + this.deps.users.insert(user); + try { + await this.deps.projectService.provisionInitialProject(user, false); + } catch (err) { + // Compensation: roll back the user row if default Project creation fails (e.g. proj- already taken). + this.deps.users.delete(user.userId); + throw err; + } + return toUserInfo(user); + } + + /** Reset another user's password: flags it as initial and clears all their sessions (prompts a password change on next login). */ + async resetPassword(userId: string, password: string): Promise { + if (!this.deps.users.findById(userId)) { + throw new HttpError(404, "user_not_found", `用户不存在:${userId}。`); + } + if (password.length < MIN_PASSWORD_LENGTH) { + throw new HttpError(400, "invalid_password", "密码至少 8 个字符。"); + } + this.deps.users.updatePassword(userId, await hashPassword(password), true); + this.deps.authSessions.deleteByUser(userId); + } + + /** Delete user: the built-in admin cannot be deleted; owned Projects (including data directories) are deleted along with it. */ + async deleteUser(userId: string): Promise { + const target = this.deps.users.findById(userId); + if (!target) { + throw new HttpError(404, "user_not_found", `用户不存在:${userId}。`); + } + if (target.isAdmin) { + throw new HttpError(409, "cannot_delete_admin", "内置管理员不能删除。"); + } + for (const project of this.deps.projects.listByOwner(userId)) { + await this.deps.projectService.destroyProject(project.projectId); + } + this.deps.users.delete(userId); // auth_sessions / project_members / ui_prefs cascade-deleted + } +} diff --git a/packages/server/src/services/agent-config-service.ts b/packages/server/src/services/agent-config-service.ts new file mode 100644 index 0000000..22599b3 --- /dev/null +++ b/packages/server/src/services/agent-config-service.ts @@ -0,0 +1,341 @@ +/** + * Agent config read/write (config is an editable file). + * + * system_config.yaml is edited via yaml's `parseDocument`: only the keys provided in + * the request are updated, the rest of the file (including comments) is preserved + * as-is; AGENTS.md is overwritten in full. + * The vault (agent_state/.vault.toml) is read/written via core's loadAgentVault/saveAgentVault; + * plaintext values only ever hit disk, and are always masked in responses. + */ +import fs from "node:fs/promises"; +import { parseDocument, parse as parseYaml } from "yaml"; +import { + agentsMdPath, + agentStateDir, + agentStateVersion, + VAULT_VALUE_MAX_LENGTH, + isValidVaultKey, + loadAgentVault, + saveAgentVault, + systemConfigPath, +} from "@prismshadow/penguin-core"; +import type { + MCPServerConfig, + ThinkingLevelName, + ToolDefinitionConfig, +} from "@prismshadow/penguin-core"; +import type { + AgentConfigDto, + AgentConfigUpdateRequest, + AgentModelConfigDto, + AgentCompactionConfigDto, + VaultEntryInfo, + VaultResponse, + VaultUpdateRequest, +} from "../api/types.js"; +import { HttpError } from "../http/errors.js"; +import { badRequest, optionalEnum, optionalNumber, optionalString } from "../http/validate.js"; +import { maskApiKey } from "./project-config-service.js"; + +const THINKING_LEVELS: readonly ThinkingLevelName[] = ["none", "low", "medium", "high", "xhigh"]; +const COMPACTION_MODES = ["summarize", "discard"] as const; + +function asRecord(v: unknown): Record { + return v !== null && typeof v === "object" && !Array.isArray(v) + ? (v as Record) + : {}; +} + +export interface AgentConfigView { + agentsMd: string; + systemConfigYaml: string; + config: AgentConfigDto; + stateDir: string; +} + +export class AgentConfigService { + constructor(private readonly root: string) {} + + /** Whether the Agent exists (determined by the presence of system_config.yaml, matching the CLI's convention). */ + async exists(projectId: string, agentId: string): Promise { + try { + await fs.access(systemConfigPath(this.root, projectId, agentId)); + return true; + } catch { + return false; + } + } + + async requireExists(projectId: string, agentId: string): Promise { + if (!(await this.exists(projectId, agentId))) { + throw new HttpError(404, "agent_not_found", "Agent 不存在。"); + } + } + + /** + * Read list-card metadata: name / description + tool count (sum of tools.builtin + * and tools.mcpServers entries; MCP counted per server). Silently falls back to + * empty / 0 if the file is corrupt. + */ + async readCardMeta( + projectId: string, + agentId: string, + ): Promise<{ name?: string; description?: string; toolCount: number; version: number }> { + try { + const raw = await fs.readFile(systemConfigPath(this.root, projectId, agentId), "utf8"); + const parsed = asRecord(parseYaml(raw)); + const tools = asRecord(parsed.tools); + const countOf = (v: unknown): number => (Array.isArray(v) ? v.length : 0); + return { + ...(typeof parsed.name === "string" ? { name: parsed.name } : {}), + ...(typeof parsed.description === "string" ? { description: parsed.description } : {}), + toolCount: countOf(tools.builtin) + countOf(tools.mcpServers), + version: agentStateVersion({ version: parsed.version as number | undefined }), + }; + } catch { + return { toolCount: 0, version: 1 }; + } + } + + /** Structured config view (matching the edit form's shape) + raw text + AGENTS.md + State path. */ + async getConfig(projectId: string, agentId: string): Promise { + await this.requireExists(projectId, agentId); + const yamlPath = systemConfigPath(this.root, projectId, agentId); + const systemConfigYaml = await fs.readFile(yamlPath, "utf8"); + const parsed = asRecord(parseYaml(systemConfigYaml)); + const model = asRecord(parsed.model); + const compaction = asRecord(parsed.compaction); + const tools = asRecord(parsed.tools); + + let agentsMd = ""; + try { + agentsMd = await fs.readFile(agentsMdPath(this.root, projectId, agentId), "utf8"); + } catch { + // Treat a missing AGENTS.md as an empty file (it normally exists after initialization). + } + + const modelDto: AgentModelConfigDto = { + ...(typeof model.max_tokens === "number" ? { maxTokens: model.max_tokens } : {}), + ...(typeof model.thinking_level === "string" + ? { thinkingLevel: model.thinking_level as ThinkingLevelName } + : {}), + ...(typeof model.timeoutMs === "number" ? { timeoutMs: model.timeoutMs } : {}), + }; + const compactionDto: AgentCompactionConfigDto = { + ...(typeof compaction.max_context_length === "number" + ? { maxContextLength: compaction.max_context_length } + : {}), + ...(typeof compaction.max_session_turns === "number" + ? { maxSessionTurns: compaction.max_session_turns } + : {}), + ...(compaction.mode === "summarize" || compaction.mode === "discard" + ? { mode: compaction.mode } + : {}), + ...(typeof compaction.prompt === "string" ? { prompt: compaction.prompt } : {}), + }; + const config: AgentConfigDto = { + ...(typeof parsed.name === "string" ? { name: parsed.name } : {}), + ...(typeof parsed.description === "string" ? { description: parsed.description } : {}), + version: agentStateVersion({ version: parsed.version as number | undefined }), + systemPrompt: typeof parsed.system_prompt === "string" ? parsed.system_prompt : "", + ...(typeof parsed.max_turns === "number" ? { maxTurns: parsed.max_turns } : {}), + ...(Object.keys(modelDto).length > 0 ? { model: modelDto } : {}), + ...(Object.keys(compactionDto).length > 0 ? { compaction: compactionDto } : {}), + toolsBuiltin: Array.isArray(tools.builtin) ? (tools.builtin as ToolDefinitionConfig[]) : [], + mcpServers: Array.isArray(tools.mcpServers) ? (tools.mcpServers as MCPServerConfig[]) : [], + }; + return { + agentsMd, + systemConfigYaml, + config, + stateDir: agentStateDir(this.root, projectId, agentId), + }; + } + + /** + * PUT accepts any subset: only the provided keys are updated (parseDocument + * preserves comments and untouched content); agentsMd is overwritten in full. + * Numeric validation: >0 or -1; thinkingLevel / mode are validated as enums. + */ + async updateConfig( + projectId: string, + agentId: string, + req: AgentConfigUpdateRequest, + ): Promise { + await this.requireExists(projectId, agentId); + // Finish all config validation and document changes before writing to disk + // (if validation fails, AGENTS.md is not written either, avoiding a partial update). + if (req.config !== undefined) { + await this.applyConfigUpdate(projectId, agentId, req.config); + } + if (req.agentsMd !== undefined) { + await fs.writeFile(agentsMdPath(this.root, projectId, agentId), req.agentsMd, "utf8"); + } + } + + private async applyConfigUpdate( + projectId: string, + agentId: string, + config: NonNullable, + ): Promise { + const cfg = config as unknown as Record; + const yamlPath = systemConfigPath(this.root, projectId, agentId); + const doc = parseDocument(await fs.readFile(yamlPath, "utf8")); + + const setIfProvided = (path: string[], value: unknown): void => { + if (value !== undefined) doc.setIn(path, value); + }; + + setIfProvided(["name"], optionalString(cfg, "name", { maxLen: 100, label: "name" })); + setIfProvided( + ["description"], + optionalString(cfg, "description", { maxLen: 2000, label: "description" }), + ); + setIfProvided( + ["system_prompt"], + optionalString(cfg, "systemPrompt", { label: "systemPrompt" }), + ); + setIfProvided( + ["max_turns"], + optionalNumber(cfg, "maxTurns", { integer: true, positiveOrMinusOne: true }), + ); + + if (cfg.model !== undefined) { + const model = asRecord(cfg.model); + setIfProvided( + ["model", "max_tokens"], + optionalNumber(model, "maxTokens", { integer: true, positiveOrMinusOne: true }), + ); + setIfProvided( + ["model", "thinking_level"], + optionalEnum(model, "thinkingLevel", THINKING_LEVELS), + ); + setIfProvided( + ["model", "timeoutMs"], + optionalNumber(model, "timeoutMs", { integer: true, positiveOrMinusOne: true }), + ); + } + if (cfg.compaction !== undefined) { + const compaction = asRecord(cfg.compaction); + setIfProvided( + ["compaction", "max_context_length"], + optionalNumber(compaction, "maxContextLength", { integer: true, positiveOrMinusOne: true }), + ); + setIfProvided( + ["compaction", "max_session_turns"], + optionalNumber(compaction, "maxSessionTurns", { integer: true, positiveOrMinusOne: true }), + ); + setIfProvided(["compaction", "mode"], optionalEnum(compaction, "mode", COMPACTION_MODES)); + setIfProvided(["compaction", "prompt"], optionalString(compaction, "prompt")); + } + if (cfg.toolsBuiltin !== undefined) { + doc.setIn(["tools", "builtin"], validateToolsBuiltin(cfg.toolsBuiltin)); + } + if (cfg.mcpServers !== undefined) { + doc.setIn(["tools", "mcpServers"], validateMcpServers(cfg.mcpServers)); + } + + await fs.writeFile(yamlPath, doc.toString(), "utf8"); + } + + /** Read the Agent vault (agent_state/.vault.toml): values are always masked, plaintext is never sent to the client. */ + async getVault(projectId: string, agentId: string): Promise { + await this.requireExists(projectId, agentId); + const vault = await loadAgentVault(this.root, projectId, agentId); + const entries: VaultEntryInfo[] = Object.entries(vault).map(([key, value]) => ({ + key, + valueMasked: maskApiKey(value), + })); + return { entries }; + } + + /** + * PUT replaces the whole vault table (same semantics as models): keys absent from + * the body are deleted; omitting value keeps the existing value (a new key must + * provide a value). Key names are validated against shell environment variable + * naming rules (same rule as core); deleting everything removes the whole + * .vault.toml file. + */ + async updateVault( + projectId: string, + agentId: string, + req: VaultUpdateRequest, + ): Promise { + await this.requireExists(projectId, agentId); + const prev = await loadAgentVault(this.root, projectId, agentId); + + const seen = new Set(); + const nextVault: Record = {}; + for (const entry of req.entries) { + if (!isValidVaultKey(entry.key)) { + throw badRequest( + `vault 键名不合法:${entry.key}(仅字母、数字与下划线,且不能以数字开头)。`, + ); + } + if (seen.has(entry.key)) { + throw badRequest(`entries 中存在重复的键名:${entry.key}。`); + } + seen.add(entry.key); + const prevValue = prev[entry.key]; + if (entry.value !== undefined) { + // Values are injected into the child process environment: an oversized value would + // make exec spawn fail (E2BIG), so we reject it on write (same limit as core). + if (entry.value.length > VAULT_VALUE_MAX_LENGTH) { + throw badRequest(`vault 值过长:${entry.key}(上限 ${VAULT_VALUE_MAX_LENGTH} 字符)。`); + } + nextVault[entry.key] = entry.value; + } else if (prevValue !== undefined) { + nextVault[entry.key] = prevValue; + } else { + throw badRequest(`新增键 ${entry.key} 必须提供 value。`); + } + } + + await saveAgentVault(this.root, projectId, agentId, nextVault); + return this.getVault(projectId, agentId); + } +} + +function validateToolsBuiltin(value: unknown): ToolDefinitionConfig[] { + if (!Array.isArray(value)) throw badRequest("toolsBuiltin 必须是数组。"); + return value.map((item, i) => { + const t = asRecord(item); + if (typeof t.name !== "string" || t.name.length === 0) { + throw badRequest(`toolsBuiltin[${i}].name 必须是非空字符串。`); + } + if (typeof t.description !== "string") { + throw badRequest(`toolsBuiltin[${i}].description 必须是字符串。`); + } + if (t.permission !== undefined && t.permission !== "r" && t.permission !== "rw") { + throw badRequest(`toolsBuiltin[${i}].permission 必须是 r / rw 之一。`); + } + if (t.forModel !== undefined && t.forModel !== "vision" && t.forModel !== "text-only") { + throw badRequest(`toolsBuiltin[${i}].forModel 必须是 vision / text-only 之一。`); + } + optionalNumber(t, "timeoutMs", { + integer: true, + positiveOrMinusOne: true, + label: `toolsBuiltin[${i}].timeoutMs`, + }); + optionalNumber(t, "maxOutputLength", { + integer: true, + positiveOrMinusOne: true, + label: `toolsBuiltin[${i}].maxOutputLength`, + }); + return t as unknown as ToolDefinitionConfig; + }); +} + +function validateMcpServers(value: unknown): MCPServerConfig[] { + if (!Array.isArray(value)) throw badRequest("mcpServers 必须是数组。"); + return value.map((item, i) => { + const s = asRecord(item); + if (typeof s.name !== "string" || s.name.length === 0) { + throw badRequest(`mcpServers[${i}].name 必须是非空字符串。`); + } + if (s.config === null || typeof s.config !== "object" || Array.isArray(s.config)) { + throw badRequest(`mcpServers[${i}].config 必须是对象。`); + } + return s as unknown as MCPServerConfig; + }); +} diff --git a/packages/server/src/services/agent-service.ts b/packages/server/src/services/agent-service.ts new file mode 100644 index 0000000..421b244 --- /dev/null +++ b/packages/server/src/services/agent-service.ts @@ -0,0 +1,219 @@ +/** + * Agent service. + * + * The list is the union of "DB index ∪ directory scan": a subdirectory under + * `/` containing `agent_state/system_config.yaml` is treated as an Agent; + * unmanaged ones found are backfilled into the DB — this handles Agents created + * directly via the CLI. + * Create: generate agent-<8hex>, initialize Agent State via core's `createAgent`, + * then write name/description into system_config.yaml (parseDocument preserves the + * template's comments). + */ +import fs from "node:fs/promises"; +import { HttpError } from "../http/errors.js"; +import { + agentDir, + agentsDir, + agentsMdPath, + BUILTIN_AGENT_IDS, + createAgent as coreCreateAgent, + isValidId, + loadAgentVault, + scheduleDir, + systemConfigPath, +} from "@prismshadow/penguin-core"; +import type { AgentsRepo } from "../db/repos/agents.js"; +import { SEMANTIC_ID_PATTERN, SEMANTIC_ID_RULE } from "./ids.js"; +import type { AgentConfigService } from "./agent-config-service.js"; + +export interface AgentListItem { + agentId: string; + name?: string; + description?: string; + createdAt?: string; + /** Last config modification time: the later of system_config.yaml / AGENTS.md mtime. */ + updatedAt?: string; + /** Tool count: number of tools.builtin + tools.mcpServers entries (MCP counted per server). */ + toolCount: number; + /** Agent State version number (missing field treated as 1). */ + version: number; + /** Number of vault keys. */ + vaultKeyCount: number; + /** Number of scheduled tasks (count of .toml files under schedule/, including invalid ones). */ + scheduleCount: number; +} + +export class AgentService { + constructor( + private readonly root: string, + private readonly agents: AgentsRepo, + private readonly agentConfig: AgentConfigService, + ) {} + + /** Union of DB index ∪ directory scan; unmanaged directory Agents are backfilled into the DB. */ + async listAgents(projectId: string): Promise { + const known = new Map(this.agents.list(projectId).map((r) => [r.agentId, r])); + + let entries: string[] = []; + try { + const dirents = await fs.readdir(agentsDir(this.root, projectId), { withFileTypes: true }); + entries = dirents.filter((d) => d.isDirectory()).map((d) => d.name); + } catch { + // The Project's agents/ directory doesn't exist yet (no Agent directories): return from the DB index only. + } + for (const agentId of entries) { + if (known.has(agentId) || !isValidId(agentId)) continue; + const configPath = systemConfigPath(this.root, projectId, agentId); + let createdAt: string; + try { + const stat = await fs.stat(configPath); + createdAt = (stat.birthtime.getTime() > 0 ? stat.birthtime : stat.mtime).toISOString(); + } catch { + continue; // A directory without system_config.yaml is not an Agent (e.g. a temp folder) + } + const row = { projectId, agentId, createdAt }; + this.agents.insertOrIgnore(row); + known.set(agentId, row); + } + + // Meta reads and mtime stats for each Agent run in parallel (Promise.all preserves the sorted order). + const sorted = [...known.values()].sort((a, b) => + a.createdAt === b.createdAt + ? a.agentId.localeCompare(b.agentId) + : a.createdAt < b.createdAt + ? -1 + : 1, + ); + return Promise.all( + sorted.map(async (row) => { + const [meta, updatedAt, vaultKeyCount, scheduleCount] = await Promise.all([ + this.agentConfig.readCardMeta(projectId, row.agentId), + this.configUpdatedAt(projectId, row.agentId), + this.vaultKeyCount(projectId, row.agentId), + this.scheduleCount(projectId, row.agentId), + ]); + return { + agentId: row.agentId, + ...meta, + createdAt: row.createdAt, + ...(updatedAt !== undefined ? { updatedAt } : {}), + vaultKeyCount, + scheduleCount, + }; + }), + ); + } + + /** Number of vault keys (falls back to 0 on read failure). */ + private async vaultKeyCount(projectId: string, agentId: string): Promise { + try { + return Object.keys(await loadAgentVault(this.root, projectId, agentId)).length; + } catch { + return 0; + } + } + + /** Number of scheduled tasks: count of .toml files under schedule/ (0 if the directory doesn't exist). */ + private async scheduleCount(projectId: string, agentId: string): Promise { + try { + const names = await fs.readdir(scheduleDir(this.root, projectId, agentId)); + return names.filter((n) => n.endsWith(".toml")).length; + } catch { + return 0; + } + } + + /** Last config modification time: the later of system_config.yaml and AGENTS.md mtime; omitted if neither is readable. */ + private async configUpdatedAt(projectId: string, agentId: string): Promise { + const paths = [ + systemConfigPath(this.root, projectId, agentId), + agentsMdPath(this.root, projectId, agentId), + ]; + const times = await Promise.all( + paths.map(async (p) => { + try { + return (await fs.stat(p)).mtime.getTime(); + } catch { + return 0; + } + }), + ); + const max = Math.max(...times); + return max > 0 ? new Date(max).toISOString() : undefined; + } + + /** + * Delete an Agent: the sole built-in Agent + * default_agent (shared with the CLI, the default conversation Agent) cannot be + * deleted; callers must first drain any active run via manager.abortAgent. + * The directory is deleted recursively (including Trace), and the DB's + * agents/sessions index rows are removed along with it; usage records are kept + * (historical stats are unaffected). + */ + async deleteAgent(projectId: string, agentId: string): Promise { + if (BUILTIN_AGENT_IDS.includes(agentId)) { + throw new HttpError( + 409, + "cannot_delete_builtin_agent", + "内置 Agent(default_agent)随 Project 供给,不能从 Web 删除。", + ); + } + await fs.rm(agentDir(this.root, projectId, agentId), { recursive: true, force: true }); + this.agents.delete(projectId, agentId); + } + + /** + * Create an Agent: the id is chosen by the creator (a semantic id, checked for + * duplicates against both the DB and the directory within the Project — a 409 + * if taken, which naturally also blocks built-in Agent ids) → initialize State → + * write name/description (name defaults to the id). + */ + async createAgent( + projectId: string, + agentId: string, + name?: string, + description?: string, + ): Promise { + if (!SEMANTIC_ID_PATTERN.test(agentId)) { + throw new HttpError(400, "invalid_agent_id", `Agent id 须为 2~64 位:${SEMANTIC_ID_RULE}。`); + } + const taken = + this.agents.exists(projectId, agentId) || + (await fs.stat(agentDir(this.root, projectId, agentId)).then( + () => true, + () => false, + )); + if (taken) { + throw new HttpError(409, "agent_exists", `Agent id 已被占用:${agentId}。`); + } + const displayName = name ?? agentId; + await coreCreateAgent({ root: this.root, projectId, agentId }); + try { + await this.agentConfig.updateConfig(projectId, agentId, { + config: { name: displayName, ...(description !== undefined ? { description } : {}) }, + }); + } catch (err) { + // If initialization fails partway through, clean up the directory: an orphaned + // directory would make retries with this agent id 409 forever. + await fs + .rm(agentDir(this.root, projectId, agentId), { recursive: true, force: true }) + .catch(() => {}); + throw err; + } + const createdAt = new Date().toISOString(); + this.agents.insertOrIgnore({ projectId, agentId, createdAt }); + // The init template ships with a default toolset and version number; read back the actual values. + const meta = await this.agentConfig.readCardMeta(projectId, agentId); + return { + agentId, + name: displayName, + ...(description !== undefined ? { description } : {}), + createdAt, + updatedAt: createdAt, + toolCount: meta.toolCount, + version: meta.version, + vaultKeyCount: 0, + scheduleCount: 0, + }; + } +} diff --git a/packages/server/src/services/benchmark-service.ts b/packages/server/src/services/benchmark-service.ts new file mode 100644 index 0000000..a19e55e --- /dev/null +++ b/packages/server/src/services/benchmark-service.ts @@ -0,0 +1,216 @@ +/** + * Benchmark score reading (read-only display): walks `benchmarks//`, reads + * `benchmark_config.toml` (title, + * description, evaluation Model, per-case run count `runs`) and `scoreboard.yaml` + * (evaluations[], scoreboard v2: each case carries a runs array and a summary). + * Content is created and refined by benchmark_builder; the server only reads it. + * Missing or corrupt files always degrade gracefully (title falls back to the + * directory name, scores come back empty) rather than throwing. + * + * The three per-case metrics trust the file's own values; when missing they're + * computed as the average over the runs array. The old format (no runs at the + * case level, a single session_id) is parsed as a single run — the server backfills + * one run entry. + * Docs: /docs/self-improvement § "Benchmark storage". + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { parse as parseToml } from "smol-toml"; +import { parse as parseYaml } from "yaml"; +import { benchmarksDir } from "@prismshadow/penguin-core"; +import type { + BenchmarkCaseScore, + BenchmarkEvaluation, + BenchmarkRunScore, + BenchmarkSummary, + BenchmarksResponse, +} from "../api/types.js"; + +function asRecord(v: unknown): Record { + return v !== null && typeof v === "object" && !Array.isArray(v) + ? (v as Record) + : {}; +} + +function numberOr(v: unknown): number | undefined { + return typeof v === "number" && Number.isFinite(v) ? v : undefined; +} + +function stringOr(v: unknown): string | undefined { + return typeof v === "string" && v !== "" ? v : undefined; +} + +/** Shapes a single run entry: score is the minimum requirement, other fields tolerate being absent; a bad entry returns null and is dropped. */ +function toRun(v: unknown): BenchmarkRunScore | null { + const r = asRecord(v); + const score = numberOr(r.score); + if (score === undefined) return null; + const cost = numberOr(r.cost); + const durationMs = numberOr(r.duration_ms); + const sessionId = stringOr(r.session_id); + return { + score, + ...(cost !== undefined ? { cost } : {}), + ...(durationMs !== undefined ? { durationMs } : {}), + ...(sessionId !== undefined ? { sessionId } : {}), + }; +} + +/** Average of a metric across runs; undefined when there's no value at all (never forced to 0). */ +function averageOf(runs: BenchmarkRunScore[], pick: (r: BenchmarkRunScore) => number | undefined) { + const values = runs.map(pick).filter((v): v is number => v !== undefined); + if (values.length === 0) return undefined; + return values.reduce((a, b) => a + b, 0) / values.length; +} + +/** + * Shapes a case-level entry (scoreboard v2): the three metrics trust the file's own + * values, falling back to an average over runs when missing; the old format (no + * runs, a single case-level session_id) is backfilled into a single run. case and a + * score (from the file or derivable from runs) are the minimum requirement, + * otherwise the entry is dropped. + */ +function toCase(v: unknown): BenchmarkCaseScore | null { + const cr = asRecord(v); + const caseId = stringOr(cr.case); + if (caseId === undefined) return null; + const parsedRuns = Array.isArray(cr.runs) + ? cr.runs.map(toRun).filter((r): r is BenchmarkRunScore => r !== null) + : []; + const score = numberOr(cr.score) ?? averageOf(parsedRuns, (r) => r.score); + if (score === undefined) return null; + const cost = numberOr(cr.cost) ?? averageOf(parsedRuns, (r) => r.cost); + const durationMs = numberOr(cr.duration_ms) ?? averageOf(parsedRuns, (r) => r.durationMs); + const sessionId = stringOr(cr.session_id); + const runs: BenchmarkRunScore[] = + parsedRuns.length > 0 + ? parsedRuns + : [ + // The old format is parsed as a single run: the case-level values are that run's raw result. + { + score, + ...(cost !== undefined ? { cost } : {}), + ...(durationMs !== undefined ? { durationMs } : {}), + ...(sessionId !== undefined ? { sessionId } : {}), + }, + ]; + return { + case: caseId, + score, + ...(cost !== undefined ? { cost } : {}), + ...(durationMs !== undefined ? { durationMs } : {}), + ...(sessionId !== undefined ? { sessionId } : {}), + runs, + }; +} + +/** Shapes a single evaluation record: time and score are the minimum requirement, other fields (summary, etc.) tolerate being absent. */ +function toEvaluation(v: unknown): BenchmarkEvaluation | null { + const r = asRecord(v); + const time = r.time instanceof Date ? r.time.toISOString() : r.time; + const score = numberOr(r.score); + if (typeof time !== "string" || time === "" || score === undefined) return null; + const cases: BenchmarkCaseScore[] = Array.isArray(r.cases) + ? r.cases.map(toCase).filter((c): c is BenchmarkCaseScore => c !== null) + : []; + const summary = stringOr(r.summary); + // Title and body are separate: summary_title is a one-line + // conclusion, summary is the body text. + const summaryTitle = stringOr(r.summary_title); + // The Model actually used for this evaluation run (paired with provider): + // charted curves are split into series by model, each with a distinct color. + const modelId = stringOr(r.model_id); + const provider = stringOr(r.provider); + const version = numberOr(r.version); + const cost = numberOr(r.cost); + const durationMs = numberOr(r.duration_ms); + return { + time, + ...(summaryTitle !== undefined ? { summaryTitle } : {}), + ...(summary !== undefined ? { summary } : {}), + ...(modelId !== undefined ? { modelId } : {}), + ...(provider !== undefined ? { provider } : {}), + score, + ...(version !== undefined ? { version } : {}), + ...(cost !== undefined ? { cost } : {}), + ...(durationMs !== undefined ? { durationMs } : {}), + cases, + }; +} + +export class BenchmarkService { + constructor(private readonly root: string) {} + + async list(projectId: string, agentId: string): Promise { + const dir = benchmarksDir(this.root, projectId, agentId); + let items: Array<{ name: string; isDir: boolean }>; + try { + const entries = await fs.readdir(dir, { withFileTypes: true }); + items = entries.map((e) => ({ name: e.name, isDir: e.isDirectory() })); + } catch { + return { benchmarks: [] }; // Doesn't exist when unconfigured. + } + const benchmarks: BenchmarkSummary[] = []; + for (const item of items.filter((i) => i.isDir).sort((a, b) => a.name.localeCompare(b.name))) { + benchmarks.push(await this.readBenchmark(path.join(dir, item.name), item.name)); + } + return { benchmarks }; + } + + private async readBenchmark(benchDir: string, id: string): Promise { + // benchmark_config.toml: title, description, and per-case run count (falls back + // to defaults if corrupt). The model isn't part of the config — each evaluation + // carries the Model actually used for that run. + let title = id; + let description: string | undefined; + let runs: number | undefined; + try { + const config = asRecord( + parseToml(await fs.readFile(path.join(benchDir, "benchmark_config.toml"), "utf8")), + ); + if (typeof config.title === "string" && config.title !== "") title = config.title; + if (typeof config.description === "string" && config.description !== "") { + description = config.description; + } + const configRuns = numberOr(config.runs); + if (configRuns !== undefined && Number.isInteger(configRuns) && configRuns >= 1) { + runs = configRuns; + } + } catch { + // Missing or corrupt: title falls back to the directory name. + } + + // scoreboard.yaml: evaluations[] is appended over time; bad entries are dropped one by one. + let evaluations: BenchmarkEvaluation[] = []; + try { + const scoreboard = asRecord( + parseYaml(await fs.readFile(path.join(benchDir, "scoreboard.yaml"), "utf8")), + ); + if (Array.isArray(scoreboard.evaluations)) { + evaluations = scoreboard.evaluations + .map(toEvaluation) + .filter((e): e is BenchmarkEvaluation => e !== null); + } + } catch { + // No scores yet. + } + + // Case count: number of case subfolders (the statement/rubric structure isn't validated here). + let caseCount = 0; + try { + const entries = await fs.readdir(benchDir, { withFileTypes: true }); + caseCount = entries.filter((e) => e.isDirectory()).length; + } catch { + // Stays at 0. + } + + return { + id, + title, + ...(description !== undefined ? { description } : {}), + ...(runs !== undefined ? { runs } : {}), + caseCount, + evaluations, + }; + } +} diff --git a/packages/server/src/services/ids.ts b/packages/server/src/services/ids.ts new file mode 100644 index 0000000..4bdee87 --- /dev/null +++ b/packages/server/src/services/ids.ts @@ -0,0 +1,34 @@ +/** + * id rules: user_id / project_id / agent_id are semantic ids + * chosen by their creator at creation time — starting with a lowercase letter, + * containing only lowercase letters, digits, and underscores. The id doubles as the + * directory name, so keeping it all-lowercase avoids directory name collisions on + * case-insensitive filesystems (e.g. macOS) at the source. + * The hyphen is a **reserved separator**: it only appears at the join point of a + * non-admin project_id's "-" concatenation. Usernames never + * contain a hyphen, so the first hyphen is the ownership boundary — no username can + * ever be crafted to collide with another user's prefix, keeping namespaces + * non-overlapping. session_id and temporary workspace ids are still generated by + * the server (randomHex8). + */ +import { randomBytes } from "node:crypto"; + +/** 8-character lowercase hex random string (used for server-generated ids like session_id / temporary workspace). */ +export function randomHex8(): string { + return randomBytes(4).toString("hex"); +} + +/** General semantic id rule: starts with a lowercase letter, followed by lowercase letters, digits, or underscores only, 2-64 chars (no hyphen). */ +export const SEMANTIC_ID_PATTERN = /^[a-z][a-z0-9_]{1,63}$/; + +/** Username tightens the general rule to 2-32 chars: leaves headroom for the default Project id `-default_project`. */ +export const USERNAME_PATTERN = /^[a-z][a-z0-9_]{1,31}$/; + +/** Suffix segment of a non-admin project_id (after `-`): lowercase letters, digits, and underscores only. */ +export const PROJECT_SUFFIX_PATTERN = /^[a-z0-9_]+$/; + +/** Upper bound on total project_id length (username <=32 + separator + suffix). */ +export const PROJECT_ID_MAX_LENGTH = 64; + +/** Human-readable description of the rule (reused in error messages). */ +export const SEMANTIC_ID_RULE = "小写字母开头,仅小写字母、数字与下划线"; diff --git a/packages/server/src/services/project-config-service.ts b/packages/server/src/services/project-config-service.ts new file mode 100644 index 0000000..a95bf63 --- /dev/null +++ b/packages/server/src/services/project-config-service.ts @@ -0,0 +1,498 @@ +/** + * `.project_config.toml` read/write (single hidden config file). + * + * Doesn't reuse core's loadProjectConfig/saveProjectConfig (they only keep known + * fields): reads and writes the complete object directly via smol-toml, preserving + * extension fields like `name`. credential (api_key / base_url / created_at) is + * **inlined on the model entry** — there's no longer a supplementary section or + * secrets file; since the file contains secrets, it's always written with mode + * 0600. Plaintext only ever hits disk, and is always masked in responses. + * + * Model references are **fully split into separate fields**: an entry is + * stored as two independent fields, `provider` and `model_id`; the `(provider, + * model_id)` pair is the entry's unique key. `model_id` is the upstream request id, + * sent to AgentHub verbatim — string concatenation like `/` is + * forbidden everywhere in the pipeline. `default_model` / `vision_model` are `{ + * provider, model_id }` paired references (TOML tables). + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { parse as parseToml } from "smol-toml"; +import { + GenerativeModel, + catalogEntryFor, + defaultProjectConfig, + projectConfigPath, + renderProjectConfigToml, + resolveModelEnv, + userText, +} from "@prismshadow/penguin-core"; +import type { ModelRef } from "@prismshadow/penguin-core"; +import type { + ModelInfo, + ModelPricingDto, + ModelRefDto, + ModelsResponse, + ModelsUpdateRequest, + ModelTestRequest, + ModelTestResponse, +} from "../api/types.js"; +import { badRequest } from "../http/validate.js"; +import type { PricingRates } from "./usage-service.js"; + +type RawTable = Record; + +/** + * API key masking: length <=12 -> `***`, otherwise `first4…last4`; plaintext is + * never sent to the client. The 12-char threshold: `first4…last4` exposes 8 + * characters, which for a 9-12 character short secret would leak more than half of + * it, so those are masked in full instead. + */ +export function maskApiKey(key: string): string { + if (key.length <= 12) return "***"; + return `${key.slice(0, 4)}…${key.slice(-4)}`; +} + +function asTable(v: unknown): RawTable { + return v !== null && typeof v === "object" && !Array.isArray(v) ? (v as RawTable) : {}; +} + +function asArray(v: unknown): RawTable[] { + return Array.isArray(v) ? v.map(asTable) : []; +} + +function optNum(v: unknown): number | undefined { + return typeof v === "number" && Number.isFinite(v) ? v : undefined; +} + +function optStr(v: unknown): string | undefined { + return typeof v === "string" && v !== "" ? v : undefined; +} + +/** Leniently reads a paired reference table (default_model / vision_model); returns undefined on a shape mismatch (including the old string format). */ +function optRef(v: unknown): ModelRef | undefined { + const t = asTable(v); + const provider = optStr(t.provider); + const modelId = optStr(t.model_id); + return provider !== undefined && modelId !== undefined + ? { provider, model_id: modelId } + : undefined; +} + +/** Whether an entry matches a paired reference (the entry's provider / model_id fields must be strings). */ +function entryMatches(m: RawTable, provider: string, modelId: string): boolean { + return m.provider === provider && m.model_id === modelId; +} + +/** In-process Map/Set key for a paired reference (\0-separated to avoid concatenation ambiguity; never persisted, not an id format). */ +function refKey(provider: string, modelId: string): string { + return `${provider}\0${modelId}`; +} + +/** Display form of a paired reference (for error messages; display only, not a storage format). */ +function showRef(provider: string, modelId: string): string { + return `(provider=${provider}, model_id=${modelId})`; +} + +export class ProjectConfigService { + constructor(private readonly root: string) {} + + private filePath(projectId: string): string { + return projectConfigPath(this.root, projectId); + } + + /** Reads the raw TOML object; returns an empty object if the file doesn't exist (does not write to disk). */ + async readRaw(projectId: string): Promise { + let raw: string; + try { + raw = await fs.readFile(this.filePath(projectId), "utf8"); + } catch (err) { + if ((err as NodeJS.ErrnoException).code === "ENOENT") return {}; + throw err; + } + return asTable(parseToml(raw)); + } + + /** + * Writes the whole object to disk: the file inlines secrets like api_key, always + * written with mode 0600 (the `mode` option only applies at creation time, so + * chmod is used to enforce it on existing files too — matching core's + * saveProjectConfig behavior). + */ + async writeRaw(projectId: string, data: RawTable): Promise { + const file = this.filePath(projectId); + await fs.mkdir(path.dirname(file), { recursive: true }); + // Rendering goes through core's single writer: paired references become inline + // tables, models is placed last — matching the CLI's output format exactly + // (the same file should never have two formats). + await fs.writeFile(file, renderProjectConfigToml(data), { encoding: "utf8", mode: 0o600 }); + await fs.chmod(file, 0o600); + } + + /** + * Initial config for a newly created Project: display name + preset built-in + * model catalog (the default model and all preset entries, sourced from the same + * core defaultProjectConfig; a gateway model's base_url is already inlined on the + * entry, with no key); users only need to fill in an API key as needed (leave it + * blank to fall back to the provider's environment variable). + */ + async writeInitialConfig(projectId: string, name: string): Promise { + const preset = defaultProjectConfig(); + await this.writeRaw(projectId, { + name, + ...(preset.default_model !== undefined ? { default_model: preset.default_model } : {}), + models: preset.models, + }); + } + + /** + * Backfills preset models (for onboarding an existing Project, e.g. the + * `default_project` shared with the CLI when the first user is onboarded — its + * directory already existed and never went through `writeInitialConfig`, so it + * previously had no models and no default model). + * + * **Only backfills when there are no models at all**: a Project that already has + * models configured (via the CLI or edited by the user) is left as-is, and its + * other fields (name, etc.) are preserved too — existing config is never + * overwritten. + */ + async ensurePresetModels(projectId: string): Promise { + const raw = await this.readRaw(projectId); + if (asArray(raw.models).length > 0) return; + const preset = defaultProjectConfig(); + await this.writeRaw(projectId, { + ...raw, + // Also reset to the preset default_model if the existing one points at a now-deleted model, to keep the default model valid. + ...(preset.default_model !== undefined ? { default_model: preset.default_model } : {}), + models: preset.models, + }); + } + + /** Project display name (the toml's name; returns undefined if unset, the frontend falls back to displaying the id). */ + async getName(projectId: string): Promise { + const raw = await this.readRaw(projectId); + return typeof raw.name === "string" ? raw.name : undefined; + } + + /** Paired reference of the default Model; returns undefined if unconfigured (or in the old string format). */ + async getDefaultModelRef(projectId: string): Promise { + const raw = await this.readRaw(projectId); + return optRef(raw.default_model); + } + + /** Pricing lookup for usage-recorder: the current pricing for this paired reference (undefined if none -> cost is NULL). */ + async getPricing( + projectId: string, + provider: string, + modelId: string, + ): Promise { + const raw = await this.readRaw(projectId); + const entry = asArray(raw.models).find((m) => entryMatches(m, provider, modelId)); + const pricing = entry ? asTable(entry.pricing) : {}; + const cacheRead = optNum(pricing.cache_read); + const cacheWrite = optNum(pricing.cache_write); + const output = optNum(pricing.output); + if (cacheRead === undefined && cacheWrite === undefined && output === undefined) { + return undefined; + } + return { cacheRead: cacheRead ?? 0, cacheWrite: cacheWrite ?? 0, output: output ?? 0 }; + } + + /** + * Model connectivity test: the model reference `(provider, modelId)` is submitted + * as a pair in the request body; sends one minimal request using that model's + * config (optionally overridden with an unsaved apiKey / baseUrl) — no tools, no + * system prompt, thinking disabled, a tiny output cap, 20s timeout — just to see + * whether it completes normally. The model id sent to AgentHub is `modelId` + * itself (the upstream id verbatim; client_type inference follows it). + * + * Never throws: the LLM layer collapses auth/parameter/network errors into an + * `LLMOutcome`, which is translated here into ok / message. Consumes very few + * Tokens (single-digit output), and writes no Trace and records no usage. + */ + async testModel(projectId: string, req: ModelTestRequest): Promise { + const raw = await this.readRaw(projectId); + // Testable even if the model isn't in the config yet (validate before saving when adding a custom model): in that case all parameters come from the request body. + const entry = asArray(raw.models).find((m) => entryMatches(m, req.provider, req.modelId)) ?? {}; + // Always tests against the **current form draft**: checking "clear" means the saved key is not fallen back to; an explicit null base URL is treated as cleared. + const savedKey = optStr(entry.api_key); + const apiKey = req.clearApiKey ? undefined : (req.apiKey ?? savedKey); + const savedBaseUrl = optStr(entry.base_url); + const baseUrl = req.baseUrl === null ? undefined : (req.baseUrl ?? savedBaseUrl); + const clientType = req.clientType ?? optStr(entry.client_type); + + const startedAt = Date.now(); + try { + // Construction must be inside the try block: the underlying provider SDK can + // throw during **client construction** itself when a credential is missing + // (models on the OpenAI protocol need apiKey/OPENAI_API_KEY) — the whole point + // of a connectivity test is to collapse that kind of failure into + // `{ ok:false }`; if construction were outside the try, a missing-key test + // would bubble up as a 500. + const llm = new GenerativeModel({ + modelId: req.modelId, + ...(apiKey ? { apiKey } : {}), + ...(baseUrl ? { baseUrl } : {}), + ...(clientType ? { clientType } : {}), + tools: [], + thinkingLevel: "none", + maxTokens: 16, + requestTimeoutMs: 20_000, + }); + const gen = llm.streamGenerate({ newMessages: [userText("ping")] }); + for (;;) { + const step = await gen.next(); + if (step.done) { + const outcome = step.value; + if (outcome.status === "completed") + return { ok: true, latencyMs: Date.now() - startedAt }; + const detail = "message" in outcome && outcome.message ? outcome.message : outcome.status; + return { ok: false, message: String(detail).slice(0, 300) }; + } + } + } catch (err) { + // Defensive: an unexpected exception during construction/iteration (the LLM layer promises not to throw; this is a fallback). + return { + ok: false, + message: (err instanceof Error ? err.message : String(err)).slice(0, 300), + }; + } + } + + /** + * GET models view: masks credential (inline fields), flags the default Model; + * the group is the entry's `provider` field, looked up in the built-in catalog by + * the `(provider, model_id)` pair to fill in displayName / envKey (entries outside + * the catalog are treated as custom models: envKey only has a fallback for the + * openai protocol). vision follows the TOML annotation when present, otherwise + * falls back to the catalog annotation (if neither exists, the field is omitted = + * supported by default). + */ + async getModels(projectId: string): Promise { + const raw = await this.readRaw(projectId); + const defaultRef = optRef(raw.default_model); + const visionRef = optRef(raw.vision_model); + const models: ModelInfo[] = asArray(raw.models) + // An entry is valid only if both provider and model_id are strings (an entry in the old concatenated format lacks provider and is ignored). + .filter((m) => typeof m.provider === "string" && typeof m.model_id === "string") + .map((m) => { + const provider = m.provider as string; + const modelId = m.model_id as string; + const pricing = asTable(m.pricing); + const pricingDto: ModelPricingDto | undefined = + optNum(pricing.cache_read) !== undefined || + optNum(pricing.cache_write) !== undefined || + optNum(pricing.output) !== undefined + ? { + cacheRead: optNum(pricing.cache_read) ?? 0, + cacheWrite: optNum(pricing.cache_write) ?? 0, + output: optNum(pricing.output) ?? 0, + } + : undefined; + const clientType = optStr(m.client_type); + const cat = catalogEntryFor(provider, modelId); + // The env fallback is reported as-is: follows the same rule as + // AgentHub routing — an explicit client_type takes priority (the openai + // protocol reads OPENAI_*, independent of the group), otherwise it's + // auto-routed to a provider client based on model_id; an id that can't be + // routed has no fallback (no envKey, and AgentHub will reject that id). + const envKey = resolveModelEnv(modelId, clientType)?.envKey; + const vision = typeof m.vision === "boolean" ? m.vision : cat?.supportsVision; + // Display name: the explicit TOML field (user-edited) takes priority, then the built-in catalog. + const displayName = optStr(m.display_name) ?? cat?.displayName; + // credential is inlined on the entry: a credential block is emitted if either api_key or base_url is present. + const apiKey = optStr(m.api_key); + const credBaseUrl = optStr(m.base_url); + const createdAt = optStr(m.created_at); + const info: ModelInfo = { + provider, + modelId, + ...(displayName !== undefined ? { displayName } : {}), + isDefault: + defaultRef !== undefined && + defaultRef.provider === provider && + defaultRef.model_id === modelId, + ...(optNum(m.context_window) !== undefined + ? { contextWindow: optNum(m.context_window)! } + : {}), + ...(clientType ? { clientType } : {}), + ...(vision !== undefined ? { vision } : {}), + ...(envKey ? { envKey } : {}), + ...(pricingDto ? { pricing: pricingDto } : {}), + ...(apiKey !== undefined || credBaseUrl !== undefined + ? { + credential: { + ...(apiKey !== undefined ? { apiKeyMasked: maskApiKey(apiKey) } : {}), + ...(credBaseUrl !== undefined ? { baseUrl: credBaseUrl } : {}), + ...(createdAt !== undefined ? { createdAt } : {}), + }, + } + : {}), + }; + return info; + }); + const toDto = (ref: ModelRef): ModelRefDto => ({ + provider: ref.provider, + modelId: ref.model_id, + }); + return { + ...(defaultRef !== undefined ? { defaultModel: toDto(defaultRef) } : {}), + ...(visionRef !== undefined ? { visionModel: toDto(visionRef) } : {}), + models, + }; + } + + /** + * PUT replaces the whole models table: key = + * `(provider, modelId)`; model entries that no longer appear are deleted along + * with their inline credential; omitting apiKey keeps the existing value, + * providing one overwrites it and records created_at, clearApiKey clears it; + * baseUrl null clears it / omitted keeps it. A key change (either the group or + * the upstream id changes) is migrated as a pair via `renamedFrom`: credential and + * unknown fields migrate along with the base entry, and default/vision pointers + * follow. Other extension fields in the toml (name, etc.) are preserved. + */ + async updateModels(projectId: string, req: ModelsUpdateRequest): Promise { + const raw = await this.readRaw(projectId); + const prevModels = asArray(raw.models); + + const seen = new Set(); + const nextModels: RawTable[] = []; + // Rename mapping (old reference key -> new reference): default model / vision model pointers follow a key change instead of being lost on a full table replacement. + const renamed = new Map(); + for (const entry of req.models) { + const key = refKey(entry.provider, entry.modelId); + if (seen.has(key)) { + throw badRequest( + `models 中存在重复的模型引用:${showRef(entry.provider, entry.modelId)}。`, + ); + } + seen.add(key); + if ( + entry.renamedFrom !== undefined && + !( + entry.renamedFrom.provider === entry.provider && + entry.renamedFrom.modelId === entry.modelId + ) + ) { + renamed.set(refKey(entry.renamedFrom.provider, entry.renamedFrom.modelId), { + provider: entry.provider, + modelId: entry.modelId, + }); + } + + // Model entry: uses the old entry (the entry for the original reference when + // the key changed) as the base, preserving unknown fields and inline + // credential; known fields are replaced wholesale per the request (omitted + // means removed). + const prevRef = entry.renamedFrom ?? { provider: entry.provider, modelId: entry.modelId }; + const prev = prevModels.find((m) => entryMatches(m, prevRef.provider, prevRef.modelId)) ?? {}; + const next: RawTable = { ...prev, provider: entry.provider, model_id: entry.modelId }; + delete next.context_window; + delete next.client_type; + delete next.vision; + delete next.pricing; + delete next.display_name; + // Leftover key from the old concatenated format (request_model_id): defensively stripped, never written to disk again. + delete next.request_model_id; + + // Display name: **only written to disk when it differs from the built-in + // catalog (looked up by the paired reference)** — preset models keep the + // config clean, only user-edited ones (including those not found in the + // catalog) get written into the TOML. + const catNew = catalogEntryFor(entry.provider, entry.modelId); + if (entry.displayName && entry.displayName !== catNew?.displayName) { + next.display_name = entry.displayName; + } + if (entry.contextWindow !== undefined) next.context_window = entry.contextWindow; + if (entry.clientType) next.client_type = entry.clientType; + // Treated as supported by default: only written to disk when explicitly annotated (both true/false are kept; false drives a frontend blocking hint). + if (entry.vision !== undefined) next.vision = entry.vision; + if (entry.pricing !== undefined) { + next.pricing = { + unit: "usd_per_mtok", + cache_read: entry.pricing.cacheRead, + cache_write: entry.pricing.cacheWrite, + output: entry.pricing.output, + }; + } + + // credential is inlined on the entry; added/removed on top of the old value per the request (migrates automatically with the base entry when the key changes). + if (entry.clearApiKey) { + delete next.api_key; + delete next.created_at; + } + if (entry.apiKey !== undefined) { + next.api_key = entry.apiKey; + next.created_at = new Date().toISOString(); + } + if (entry.baseUrl === null) delete next.base_url; + else if (entry.baseUrl !== undefined) next.base_url = entry.baseUrl; + nextModels.push(next); + } + + // default_model: when provided it must be present in models; when omitted the previous value is kept (the pointer follows a key rename; if it was deleted, it's removed). + let defaultModel: ModelRefDto | undefined; + if (req.defaultModel !== undefined) { + if (!seen.has(refKey(req.defaultModel.provider, req.defaultModel.modelId))) { + throw badRequest( + `defaultModel 必须包含在 models 内:${showRef(req.defaultModel.provider, req.defaultModel.modelId)}。`, + ); + } + defaultModel = req.defaultModel; + } else { + const prevRef = optRef(raw.default_model); + if (prevRef !== undefined) { + const prevKey = refKey(prevRef.provider, prevRef.model_id); + const followed = renamed.get(prevKey) ?? { + provider: prevRef.provider, + modelId: prevRef.model_id, + }; + if (seen.has(refKey(followed.provider, followed.modelId))) defaultModel = followed; + } + } + + // vision_model: same semantics as default_model; additionally must not be annotated vision=false (can't proxy-read images if unsupported). + const targetOf = (ref: ModelRefDto) => + req.models.find((m) => m.provider === ref.provider && m.modelId === ref.modelId); + let visionModel: ModelRefDto | undefined; + if (req.visionModel !== undefined) { + if (!seen.has(refKey(req.visionModel.provider, req.visionModel.modelId))) { + throw badRequest( + `visionModel 必须包含在 models 内:${showRef(req.visionModel.provider, req.visionModel.modelId)}。`, + ); + } + if (targetOf(req.visionModel)?.vision === false) { + throw badRequest( + `visionModel 不能指向标注为不支持图片的模型:${showRef(req.visionModel.provider, req.visionModel.modelId)}。`, + ); + } + visionModel = req.visionModel; + } else { + const prevRef = optRef(raw.vision_model); + if (prevRef !== undefined) { + const prevKey = refKey(prevRef.provider, prevRef.model_id); + const followed = renamed.get(prevKey) ?? { + provider: prevRef.provider, + modelId: prevRef.model_id, + }; + if (seen.has(refKey(followed.provider, followed.modelId))) { + // The former vision model is now annotated as not supporting images: the annotation takes priority, and the pointer is dropped as invalid. + if (targetOf(followed)?.vision !== false) visionModel = followed; + } + } + } + + const toRaw = (ref: ModelRefDto): RawTable => ({ + provider: ref.provider, + model_id: ref.modelId, + }); + const next: RawTable = { ...raw, models: nextModels }; + if (defaultModel !== undefined) next.default_model = toRaw(defaultModel); + else delete next.default_model; + if (visionModel !== undefined) next.vision_model = toRaw(visionModel); + else delete next.vision_model; + await this.writeRaw(projectId, next); + return this.getModels(projectId); + } +} diff --git a/packages/server/src/services/project-service.ts b/packages/server/src/services/project-service.ts new file mode 100644 index 0000000..6fa230c --- /dev/null +++ b/packages/server/src/services/project-service.ts @@ -0,0 +1,350 @@ +/** + * Project service. + * + * The single implementation point for authorization rules: `requireProjectAccess` + * (owner or member, otherwise 404 without leaking existence) and + * `requireProjectOwner` (owner only; 403 when known to be accessible, 404 when not) + * are reused by every route; the non-throwing `canAccess` (for error attribution) + * is likewise just a sibling wrapper around them — all three share the single + * `resolveAccess` decision, with no second rule set maintained separately. + * Also handles Project create / list / delete, member authorization, and initial + * Project provisioning at signup. + */ +import fs from "node:fs/promises"; +import { DEFAULT_PROJECT_ID, projectDir, provisionProjectAgents } from "@prismshadow/penguin-core"; +import type { MemberInfo, ProjectRole, ProjectSummary } from "../api/types.js"; +import { HttpError } from "../http/errors.js"; +import type { AgentsRepo } from "../db/repos/agents.js"; +import type { ErrorsRepo } from "../db/repos/errors.js"; +import type { MembersRepo } from "../db/repos/members.js"; +import type { ProjectRow, ProjectsRepo } from "../db/repos/projects.js"; +import type { SessionsRepo } from "../db/repos/sessions.js"; +import type { SchedulesRepo } from "../db/repos/schedules.js"; +import type { UsageRepo } from "../db/repos/usage.js"; +import type { UserRow, UsersRepo } from "../db/repos/users.js"; +import type { SessionManager } from "../runtime/session-manager.js"; +import { + PROJECT_ID_MAX_LENGTH, + PROJECT_SUFFIX_PATTERN, + SEMANTIC_ID_PATTERN, + SEMANTIC_ID_RULE, +} from "./ids.js"; +import type { ProjectConfigService } from "./project-config-service.js"; + +/** Fallback timeout for waiting on runs to settle before deleting a Project. */ +const ABORT_SETTLE_TIMEOUT_MS = 5000; + +async function dirExists(path: string): Promise { + try { + const stat = await fs.stat(path); + return stat.isDirectory(); + } catch { + return false; + } +} + +export interface ProjectServiceDeps { + root: string; + users: UsersRepo; + projects: ProjectsRepo; + members: MembersRepo; + agents: AgentsRepo; + sessions: SessionsRepo; + usage: UsageRepo; + errors: ErrorsRepo; + schedules: SchedulesRepo; + projectConfig: ProjectConfigService; + manager: SessionManager; +} + +export class ProjectService { + constructor(private readonly deps: ProjectServiceDeps) {} + + // —— Authorization rules (single implementation point) —— + + /** + * The **sole** implementation of the owner / member check: returns the row with + * a role if accessible, otherwise null. `requireProjectAccess` below (throws 404) + * and `canAccess` (returns boolean) are both just wrappers around it — there's + * only one copy of the decision rule, since writing it twice would eventually + * drift out of sync. + */ + private resolveAccess( + userId: string, + projectId: string, + ): (ProjectRow & { role: ProjectRole }) | null { + const row = this.deps.projects.findById(projectId); + if (!row) return null; + if (row.ownerUserId === userId) return { ...row, role: "owner" }; + if (this.deps.members.isMember(projectId, userId)) return { ...row, role: "member" }; + return null; + } + + /** Accessible by owner or member; otherwise 404 (does not leak Project existence). */ + requireProjectAccess(userId: string, projectId: string): ProjectRow & { role: ProjectRole } { + const row = this.resolveAccess(userId, projectId); + if (!row) { + throw new HttpError(404, "project_not_found", "Project 不存在或无权访问。"); + } + return row; + } + + /** + * The non-throwing version of the same check: used for **error attribution** + * (app.onError) — that runs on the error-handling path, where throwing another + * 404 would only break error handling; whether access is granted shouldn't be + * expressed as an exception there anyway. + */ + canAccess(userId: string, projectId: string): boolean { + return this.resolveAccess(userId, projectId) !== null; + } + + /** Owner only: 403 when known accessible as a member; 404 when not accessible. */ + requireProjectOwner(userId: string, projectId: string): ProjectRow { + const row = this.requireProjectAccess(userId, projectId); + if (row.role !== "owner") { + throw new HttpError(403, "owner_required", "该操作仅 Project owner 可执行。"); + } + return row; + } + + /** List of project_ids accessible to the current user (owned + granted access) (used by workspace-guard). */ + accessibleProjectIds(userId: string): string[] { + return this.deps.projects.listAccessible(userId).map((p) => p.projectId); + } + + // —— Project lifecycle —— + + /** List of owned + granted-access Projects; display names are read from each project_config.toml. */ + async listProjects(userId: string): Promise { + const rows = this.deps.projects.listAccessible(userId); + return Promise.all( + rows.map(async (row) => { + const name = await this.deps.projectConfig.getName(row.projectId); + return { + projectId: row.projectId, + ...(name !== undefined ? { name } : {}), + role: row.role, + ownerUserId: row.ownerUserId, + createdAt: row.createdAt, + }; + }), + ); + } + + /** + * Create a Project: the id is chosen by the creator (a semantic id, checked for + * duplicates against both the DB and the directory — 409 if taken), the initial + * config is written (display name defaults to the id), and the built-in Agent is + * initialized. + * A non-admin's id is forced to be "-", where the suffix is + * lowercase letters, digits, and underscores only — the hyphen is a reserved + * separator, usernames never contain a hyphen, so the first hyphen is the + * ownership boundary and the prefix can never be crafted from another username; + * an admin's id contains no hyphen (occupying no user's namespace). + */ + async createProject(owner: UserRow, projectId: string, name?: string): Promise { + if (owner.isAdmin) { + if (!SEMANTIC_ID_PATTERN.test(projectId)) { + throw new HttpError( + 400, + "invalid_project_id", + `Project id 须为 2~64 位:${SEMANTIC_ID_RULE}(连字符保留作用户命名空间分隔)。`, + ); + } + } else { + const prefix = `${owner.userId}-`; + const suffix = projectId.startsWith(prefix) ? projectId.slice(prefix.length) : ""; + if (!PROJECT_SUFFIX_PATTERN.test(suffix) || projectId.length > PROJECT_ID_MAX_LENGTH) { + throw new HttpError( + 400, + "project_id_prefix_required", + `Project id 须以 ${prefix} 开头,后接小写字母、数字或下划线(总长不超过 ${PROJECT_ID_MAX_LENGTH})。`, + ); + } + } + if ( + this.deps.projects.findById(projectId) !== null || + (await dirExists(projectDir(this.deps.root, projectId))) + ) { + throw new HttpError(409, "project_exists", `Project id 已被占用:${projectId}。`); + } + const displayName = name ?? projectId; + const createdAt = new Date().toISOString(); + // Insert the DB row first: the primary key is the final arbiter for concurrent + // creation with the same id (the duplicate check above has an await gap), and a + // conflict is mapped to 409 with **no cleanup** — the directory belongs to the + // winner, and cleaning up here would wrongly delete the other side's data. + try { + this.deps.projects.insert({ projectId, ownerUserId: owner.userId, createdAt }); + } catch (err) { + if (err instanceof Error && err.message.includes("UNIQUE")) { + throw new HttpError(409, "project_exists", `Project id 已被占用:${projectId}。`); + } + throw err; + } + // If file initialization fails, roll back the DB row and clean up the + // directory: an orphaned directory would make retries with this id 409 forever + // (a typical scenario: signup failure rolled back the user row, but the + // -default_project directory was left behind). + try { + await fs.mkdir(projectDir(this.deps.root, projectId), { recursive: true }); + await this.deps.projectConfig.writeInitialConfig(projectId, displayName); + await this.provisionBuiltinAgents(projectId); + } catch (err) { + this.deps.projects.delete(projectId); + await fs + .rm(projectDir(this.deps.root, projectId), { recursive: true, force: true }) + .catch(() => {}); + throw err; + } + return { + projectId, + name: displayName, + role: "owner", + ownerUserId: owner.userId, + createdAt, + }; + } + + /** + * Initial Project provisioned at signup: + * the built-in admin adopts `default_project` (if the directory already exists, + * it's adopted directly without overwriting existing config — shared with the + * CLI); other users get `-default_project` created, with display name + * defaulting to the username. + */ + async provisionInitialProject(user: UserRow, isAdmin: boolean): Promise { + if (!isAdmin) { + await this.createProject(user, `${user.userId}-${DEFAULT_PROJECT_ID}`, user.userId); + return; + } + const projectId = DEFAULT_PROJECT_ID; + // Initialize the built-in Agent (loaded without overwriting if it already exists); this also ensures the directory exists. + await this.provisionBuiltinAgents(projectId); + // Adopting an existing directory doesn't go through writeInitialConfig: preset + // models and the default model are backfilled instead (only when there are no + // models at all; a default_project already configured via the CLI is left + // as-is). + await this.deps.projectConfig.ensurePresetModels(projectId); + this.deps.projects.insert({ + projectId, + ownerUserId: user.userId, + createdAt: new Date().toISOString(), + }); + } + + /** + * Delete a Project (owner): default_project is refused; deleting the user's + * **last accessible Project** is refused too (deleting it would leave the list + * empty, with no Project to select in the Web client and the page stuck on a + * skeleton screen — a typical case being a non-first user deleting the initial + * Project provisioned at signup); active runs are drained first, then the DB and + * directory are cleared. + */ + async deleteProject(userId: string, projectId: string): Promise { + this.requireProjectOwner(userId, projectId); + if (projectId === DEFAULT_PROJECT_ID) { + throw new HttpError( + 409, + "cannot_delete_default_project", + "default_project 与 CLI 共用,不能从 Web 删除。", + ); + } + if (this.deps.projects.listAccessible(userId).length <= 1) { + throw new HttpError( + 409, + "cannot_delete_last_project", + "这是当前账号最后一个 Project,删除后将无 Project 可用;请先创建新的 Project。", + ); + } + await this.destroyProject(projectId); + } + + /** + * The actual deletion (no authorization or protection checks): shared by + * deleteProject and the cascade cleanup when an admin deletes a user. + * Abort follow-up (writing the abort event to Trace, etc.) happens + * asynchronously: waits for runs to settle (capped at 5s) before deleting the + * directory, to avoid the Trace writer recreating the directory after deletion. + */ + async destroyProject(projectId: string): Promise { + const runnings = this.deps.manager.abortProject(projectId); + if (runnings.length > 0) { + await Promise.race([ + Promise.allSettled(runnings).then(() => undefined), + new Promise((resolve) => setTimeout(resolve, ABORT_SETTLE_TIMEOUT_MS).unref?.()), + ]); + } + this.deps.projects.delete(projectId); // project_members cascade-deleted + this.deps.agents.deleteByProject(projectId); + this.deps.sessions.deleteByProject(projectId); + this.deps.usage.deleteByProject(projectId); + this.deps.errors.deleteByProject(projectId); + this.deps.schedules.deleteByProject(projectId); + await fs.rm(projectDir(this.deps.root, projectId), { recursive: true, force: true }); + } + + // —— Member authorization —— + + /** Member list: owner (role=owner) + members. */ + listMembers(userId: string, projectId: string): MemberInfo[] { + const project = this.requireProjectAccess(userId, projectId); + const members = this.deps.members.list(projectId); + return [ + { userId: project.ownerUserId, role: "owner", createdAt: project.createdAt }, + ...members.map((m) => ({ + userId: m.userId, + role: "member" as const, + createdAt: m.createdAt, + })), + ]; + } + + /** Grant member access (owner): invites by username; 404 if the user doesn't exist, 409 for the owner themself or an existing member. */ + addMember(userId: string, projectId: string, targetUserId: string): MemberInfo { + const project = this.requireProjectOwner(userId, projectId); + const target = this.deps.users.findById(targetUserId); + if (!target) { + throw new HttpError(404, "user_not_found", `用户不存在:${targetUserId}。`); + } + if (target.userId === project.ownerUserId) { + throw new HttpError(409, "already_owner", "owner 无需授权给自己。"); + } + if (this.deps.members.isMember(projectId, target.userId)) { + throw new HttpError(409, "already_member", `${targetUserId} 已是该 Project 的成员。`); + } + const createdAt = new Date().toISOString(); + this.deps.members.insert({ projectId, userId: target.userId, createdAt }); + return { userId: target.userId, role: "member", createdAt }; + } + + /** Revoke member access (owner). */ + removeMember(userId: string, projectId: string, targetUserId: string): void { + this.requireProjectOwner(userId, projectId); + if (!this.deps.members.isMember(projectId, targetUserId)) { + throw new HttpError(404, "member_not_found", `该 Project 没有成员:${targetUserId}。`); + } + this.deps.members.delete(projectId, targetUserId); + } + + /** + * Ensures the Project's built-in Agent exists (the sole built-in Agent + * default_agent; initialized if the directory is empty, otherwise loaded without + * overwriting) and indexes it. createdAt increments by 1ms in preset order, so + * built-in Agents stably sort first; other Agents backfilled by directory + * scanning are sorted by their own createdAt and are outside the scope of this + * guarantee. + */ + private async provisionBuiltinAgents(projectId: string): Promise { + const agentIds = await provisionProjectAgents({ root: this.deps.root, projectId }); + const base = Date.now(); + agentIds.forEach((agentId, i) => { + this.deps.agents.insertOrIgnore({ + projectId, + agentId, + createdAt: new Date(base + i).toISOString(), + }); + }); + } +} diff --git a/packages/server/src/services/session-service.ts b/packages/server/src/services/session-service.ts new file mode 100644 index 0000000..418bac8 --- /dev/null +++ b/packages/server/src/services/session-service.ts @@ -0,0 +1,302 @@ +/** + * Session index service. + * + * The list is DB index ∪ Trace directory discovery: scans + * `/traces//_.jsonl`; an unmanaged Session (e.g. + * one started via the CLI) has its first line's session_meta read for + * (provider, model_id) / workspace, which is backfilled into a DB row + * (approval_mode defaults, createdAt is taken from the timestamp embedded in + * session_id). + * Create: via core's `agent.createSession` (model reference as a provider + modelId + * pair; defaults to the Project's default reference, 400 if there is none; omitting + * provider goes through resolveModelRef for unique resolution); the new Session is + * added to session-manager's active table (state idle). + */ +import path from "node:path"; +import { readdir } from "node:fs/promises"; +import { + createAgent, + isSessionMeta, + readTraceTolerant, + tracesDir, +} from "@prismshadow/penguin-core"; +import type { ApprovalMode, SessionInfo } from "../api/types.js"; +import { HttpError, isMissingCredential, modelCredentialMissing } from "../http/errors.js"; +import type { SessionRow, SessionsRepo } from "../db/repos/sessions.js"; +import type { SessionManager } from "../runtime/session-manager.js"; +import type { ProjectConfigService } from "./project-config-service.js"; + +const TRACE_FILE_RE = /^(.+)_(\d{3})\.jsonl$/; +const SESSION_ID_TS_RE = /^session-(\d{4})-(\d{2})-(\d{2})-(\d{2})-(\d{2})-(\d{2})-[0-9a-f]{8}$/; + +/** Derives creation time from the local timestamp embedded in session_id; returns null if it doesn't match. */ +export function sessionIdCreatedAt(sessionId: string): string | null { + const m = SESSION_ID_TS_RE.exec(sessionId); + if (!m) return null; + const [, y, mo, d, h, mi, s] = m; + const date = new Date(Number(y), Number(mo) - 1, Number(d), Number(h), Number(mi), Number(s)); + return Number.isNaN(date.getTime()) ? null : date.toISOString(); +} + +export interface SessionServiceDeps { + root: string; + sessions: SessionsRepo; + manager: SessionManager; + projectConfig: ProjectConfigService; +} + +export class SessionService { + constructor(private readonly deps: SessionServiceDeps) {} + + /** DB row -> SessionInfo (run status and pending approval count come from session-manager). */ + toInfo(row: SessionRow, hasTrace: boolean): SessionInfo { + return { + sessionId: row.sessionId, + projectId: row.projectId, + agentId: row.agentId, + provider: row.provider, + modelId: row.modelId, + workspace: row.workspace, + approvalMode: row.approvalMode, + ...(row.title !== null ? { title: row.title } : {}), + ...(row.source != null ? { source: row.source } : {}), + createdAt: row.createdAt, + status: this.deps.manager.statusOf(row.sessionId), + pendingApprovalCount: this.deps.manager.pendingApprovalCount(row.sessionId), + hasTrace, + archived: (row.archivedAt ?? null) !== null, + }; + } + + /** Whether this Session already has a Trace record (a Task has been run). */ + async hasTrace(row: SessionRow): Promise { + const ids = await this.discoverTraceSessionIds(row.projectId, row.agentId); + return ids.has(row.sessionId); + } + + /** List: DB ∪ Trace directory discovery, sorted by createdAt descending. */ + async listSessions(projectId: string, agentId: string): Promise { + const traceIds = await this.discoverTraceSessionIds(projectId, agentId); + const rows = new Map( + this.deps.sessions.listByAgent(projectId, agentId).map((r) => [r.sessionId, r]), + ); + + // Unmanaged Trace Sessions: backfill an index row by reading the first line's session_meta. + for (const sessionId of traceIds) { + if (rows.has(sessionId)) continue; + const discovered = await this.adoptTraceSession(projectId, agentId, sessionId); + if (discovered) rows.set(sessionId, discovered); + } + + return [...rows.values()] + .sort( + (a, b) => b.createdAt.localeCompare(a.createdAt) || b.sessionId.localeCompare(a.sessionId), + ) + .map((row) => this.toInfo(row, traceIds.has(row.sessionId))); + } + + /** + * Session stats (Agents list card): total count = size of the union of DB index + * ∪ Trace directory discovery; activity = number of active Sessions per day over + * the last `days` days (deduplicated count of Sessions created that day or with a + * Trace record that day; index 0 = earliest, last index = today). Counts only — + * does not backfill index rows. + */ + async sessionStats( + projectId: string, + agentId: string, + days: number, + ): Promise<{ sessionCount: number; activity: number[] }> { + const all = new Set(); + const byDate = new Map>(); + const mark = (date: string, sessionId: string): void => { + all.add(sessionId); + const set = byDate.get(date) ?? new Set(); + set.add(sessionId); + byDate.set(date, set); + }; + + // Trace directory: the date directory name is the local date (yyyy-mm-dd) that core uses when writing to disk. + const dir = tracesDir(this.deps.root, projectId, agentId); + for (const dateDir of await listDirsSafe(dir)) { + for (const file of await listFilesSafe(path.join(dir, dateDir))) { + const match = TRACE_FILE_RE.exec(file); + if (match) mark(dateDir, match[1]!); + } + } + // DB index: the creation day also counts as active (a Session that hasn't run a Task yet produces no Trace). + for (const row of this.deps.sessions.listByAgent(projectId, agentId)) { + const created = new Date(row.createdAt); + if (Number.isNaN(created.getTime())) all.add(row.sessionId); + else mark(localDate(created), row.sessionId); + } + + const activity: number[] = []; + const now = new Date(); + for (let i = days - 1; i >= 0; i--) { + const d = new Date(now.getFullYear(), now.getMonth(), now.getDate() - i); + activity.push(byDate.get(localDate(d))?.size ?? 0); + } + return { sessionCount: all.size, activity }; + } + + /** + * Create a Session: model reference `(provider, modelId)` as a pair; defaults to + * the Project's default reference (400 prompting to configure a model first if + * there is none); when provider is omitted, core's resolveModelRef performs + * unique resolution (400 on zero matches / ambiguity). `workspace` is already + * validated by the route guard. The new Session is added to the active table + * (idle). + */ + async createSession(args: { + projectId: string; + agentId: string; + /** Upstream id of the session's model (paired with provider); defaults to the Project's default reference. */ + modelId?: string; + /** The provider group for `modelId`; if omitted, resolveModelRef performs unique resolution. */ + provider?: string; + workspace?: string; + approvalMode?: ApprovalMode; + /** Session source marker (schedule when triggered by a scheduled task; defaults to user-created). */ + source?: "schedule"; + }): Promise { + let modelId = args.modelId; + let provider = args.provider; + if (modelId === undefined) { + const def = await this.deps.projectConfig.getDefaultModelRef(args.projectId); + if (def === undefined) { + throw new HttpError( + 400, + "no_default_model", + "该 Project 尚未配置默认模型,请先在「模型」页添加模型并设为默认。", + ); + } + modelId = def.model_id; + provider = def.provider; + } + const agent = await createAgent({ + root: this.deps.root, + projectId: args.projectId, + agentId: args.agentId, + }); + let session; + try { + session = await agent.createSession({ + modelId, + ...(provider !== undefined ? { provider } : {}), + ...(args.workspace !== undefined ? { workspaceDir: args.workspace } : {}), + }); + } catch (err) { + // A missing credential is its own category (the frontend shows localized text + // by code); other core errors (zero matches / ambiguous reference, Workspace + // not existing, etc.) are collapsed to 400 — the guard already blocks most cases. + if (isMissingCredential(err)) throw modelCredentialMissing(modelId); + throw new HttpError( + 400, + "session_create_failed", + err instanceof Error ? err.message : String(err), + ); + } + const row: SessionRow = { + sessionId: session.sessionId, + projectId: args.projectId, + agentId: args.agentId, + provider: session.provider, + modelId: session.modelId, + workspace: session.workspaceDir, + approvalMode: args.approvalMode ?? "allow-all", + title: null, + createdAt: new Date().toISOString(), + ...(args.source !== undefined ? { source: args.source } : {}), + }; + this.deps.sessions.insert(row); + this.deps.manager.adopt(row, session); + return this.toInfo(row, false); + } + + /** Scans the Trace directory to get the set of session_ids with records. */ + private async discoverTraceSessionIds(projectId: string, agentId: string): Promise> { + const dir = tracesDir(this.deps.root, projectId, agentId); + const ids = new Set(); + for (const dateDir of await listDirsSafe(dir)) { + for (const file of await listFilesSafe(path.join(dir, dateDir))) { + const match = TRACE_FILE_RE.exec(file); + if (match) ids.add(match[1]!); + } + } + return ids; + } + + /** Adopts a Session that exists only in the Trace directory: reads session_meta from the first line of the earliest index file. */ + private async adoptTraceSession( + projectId: string, + agentId: string, + sessionId: string, + ): Promise { + const dir = tracesDir(this.deps.root, projectId, agentId); + let earliest: { path: string; index: number } | null = null; + for (const dateDir of await listDirsSafe(dir)) { + for (const file of await listFilesSafe(path.join(dir, dateDir))) { + const match = TRACE_FILE_RE.exec(file); + if (!match || match[1] !== sessionId) continue; + const index = Number(match[2]); + if (!earliest || index < earliest.index) { + earliest = { path: path.join(dir, dateDir, file), index }; + } + } + } + if (!earliest) return null; + let messages; + try { + messages = await readTraceTolerant(earliest.path); + } catch { + return null; // Corrupt file: skip (does not block the list) + } + const meta = messages.find(isSessionMeta); + if (!meta) return null; + // An older Trace version's session_meta lacks provider (the model reference + // wasn't split into separate fields yet): no backward compat, skip adoption + // (core will give a clear error on resume; the product hasn't launched yet, so + // old data can simply be deleted and recreated). + if (typeof meta.payload.provider !== "string") return null; + const row: SessionRow = { + sessionId, + projectId, + agentId, + provider: meta.payload.provider, + modelId: meta.payload.model_id, + workspace: meta.payload.workspace, + // The approval mode for an unmanaged Session (started via the CLI) isn't in the Trace, so it's backfilled with the default value. + approvalMode: "allow-all", + title: null, + createdAt: sessionIdCreatedAt(sessionId) ?? meta.timestamp, + }; + // Idempotent backfill: concurrent list calls may discover the same Session for the first time simultaneously (consistent with AgentsRepo's convention). + this.deps.sessions.insertOrIgnore(row); + return row; + } +} + +/** Local date as yyyy-mm-dd (matches the Trace date directory convention: core's internal formatLocalDate, not publicly exported). */ +function localDate(d: Date): string { + const pad = (n: number) => (n < 10 ? `0${n}` : `${n}`); + return `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())}`; +} + +async function listDirsSafe(dir: string): Promise { + try { + const entries = await readdir(dir, { withFileTypes: true }); + return entries.filter((e) => e.isDirectory()).map((e) => e.name); + } catch { + return []; + } +} + +async function listFilesSafe(dir: string): Promise { + try { + const entries = await readdir(dir, { withFileTypes: true }); + return entries.filter((e) => e.isFile()).map((e) => e.name); + } catch { + return []; + } +} diff --git a/packages/server/src/services/snapshot-service.ts b/packages/server/src/services/snapshot-service.ts new file mode 100644 index 0000000..bde969b --- /dev/null +++ b/packages/server/src/services/snapshot-service.ts @@ -0,0 +1,190 @@ +/** + * Agent State version snapshots and export/import. + * + * A snapshot = `snapshots/v.tar.gz`, packaging `agent_state/` (archive + * entries are rooted at `agent_state/`), **excluding `.vault.toml`** (secrets never + * go into a snapshot); if a snapshot for the same version already exists, it isn't + * repacked. Import goes by the `version` inside the package and keeps the current + * vault; a snapshot of the current version is automatically taken before import; + * importing a package version equal to or lower than the current version requires + * explicit confirmation (otherwise 409). + * Docs: /docs/self-improvement § "Snapshots and versions". + */ +import { randomBytes } from "node:crypto"; +import fs from "node:fs/promises"; +import path from "node:path"; +import * as tar from "tar"; +import { parse as parseYaml } from "yaml"; +import { + agentDir, + agentStateDir, + agentStateVersion, + agentVaultPath, + snapshotsDir, + systemConfigPath, +} from "@prismshadow/penguin-core"; +import { HttpError } from "../http/errors.js"; +import { badRequest } from "../http/validate.js"; + +/** Vault file name inside a snapshot/import package (used for archive filtering). */ +const VAULT_BASENAME = ".vault.toml"; + +function isVaultEntry(entryPath: string): boolean { + return path.posix.basename(entryPath.replaceAll("\\", "/")) === VAULT_BASENAME; +} + +export class SnapshotService { + constructor(private readonly root: string) {} + + /** Current Agent State version number (missing field treated as 1); throws 404 if the Agent doesn't exist. */ + async currentVersion(projectId: string, agentId: string): Promise { + let raw: string; + try { + raw = await fs.readFile(systemConfigPath(this.root, projectId, agentId), "utf8"); + } catch { + throw new HttpError(404, "agent_not_found", "Agent 不存在。"); + } + const parsed = parseYaml(raw) as { version?: unknown } | null; + return agentStateVersion({ + version: typeof parsed?.version === "number" ? parsed.version : undefined, + }); + } + + /** Ensures a snapshot exists for the current version (not repacked for the same version), returns the snapshot file path and version number. */ + async ensureSnapshot( + projectId: string, + agentId: string, + ): Promise<{ version: number; file: string }> { + const version = await this.currentVersion(projectId, agentId); + const dir = snapshotsDir(this.root, projectId, agentId); + const file = path.join(dir, `v${version}.tar.gz`); + try { + await fs.access(file); + return { version, file }; + } catch { + // No snapshot for this version yet: pack it. + } + await fs.mkdir(dir, { recursive: true }); + const tmp = `${file}.tmp-${randomBytes(4).toString("hex")}`; + await tar.create( + { + gzip: true, + cwd: agentDir(this.root, projectId, agentId), + file: tmp, + portable: true, + filter: (p) => !isVaultEntry(p), + }, + ["agent_state"], + ); + await fs.rename(tmp, file); + return { version, file }; + } + + /** Export: automatically packs a snapshot first if none exists, returns the info needed for download. */ + async exportArchive( + projectId: string, + agentId: string, + ): Promise<{ version: number; file: string; fileName: string }> { + const { version, file } = await this.ensureSnapshot(projectId, agentId); + return { version, file, fileName: `${agentId}-v${version}.tar.gz` }; + } + + /** + * Import: validates the package structure and `version`, compares versions + * (higher than current imports directly, same version or older requires + * `confirm`), automatically snapshots the current version before import, then + * replaces `agent_state/` while keeping the current vault. + */ + async importArchive( + projectId: string, + agentId: string, + archive: Buffer, + confirm: boolean, + ): Promise<{ version: number }> { + const current = await this.currentVersion(projectId, agentId); + const base = agentDir(this.root, projectId, agentId); + const staging = path.join(base, `.import-${randomBytes(6).toString("hex")}`); + await fs.mkdir(staging, { recursive: true }); + try { + const archiveFile = path.join(staging, "archive.tar.gz"); + await fs.writeFile(archiveFile, archive); + const extractDir = path.join(staging, "extracted"); + await fs.mkdir(extractDir, { recursive: true }); + try { + await tar.extract({ + file: archiveFile, + cwd: extractDir, + filter: (p) => !isVaultEntry(p), + }); + } catch { + throw badRequest("导入失败:不是合法的 tar.gz 快照包。"); + } + + // Validation: the package must contain agent_state/system_config.yaml, and version must be valid. + const configPath = path.join(extractDir, "agent_state", "system_config.yaml"); + let parsed: unknown; + try { + parsed = parseYaml(await fs.readFile(configPath, "utf8")); + } catch { + throw badRequest("导入失败:包内缺少 agent_state/system_config.yaml。"); + } + if ( + parsed === null || + typeof parsed !== "object" || + typeof (parsed as { system_prompt?: unknown }).system_prompt !== "string" + ) { + throw badRequest("导入失败:包内 system_config.yaml 非法。"); + } + const incomingRaw = (parsed as { version?: unknown }).version; + if ( + incomingRaw !== undefined && + (!Number.isInteger(incomingRaw) || (incomingRaw as number) < 1) + ) { + throw badRequest("导入失败:包内 version 非法。"); + } + const incoming = agentStateVersion({ + version: typeof incomingRaw === "number" ? incomingRaw : undefined, + }); + if (incoming <= current && !confirm) { + throw new HttpError( + 409, + "version_conflict", + `包版本 v${incoming} 不高于当前 v${current},需要确认后导入。`, + ); + } + + // Automatically snapshots the current version before import (reused if one already exists for that version), so a mistaken import can be rolled back. + await this.ensureSnapshot(projectId, agentId); + + // Replace agent_state: first merge the current vault into the staging + // directory to be swapped in (extraction already filtered out any vault + // inside the package, so no conflict), making the replacement a pure rename + // swap — once the swap lands, there's no further write that can fail. The + // vault never goes into a snapshot, so if recovery fails after the swap, the + // `finally` cleanup of staging would delete the only vault copy with no way + // to roll back. + const stateDir = agentStateDir(this.root, projectId, agentId); + const incomingState = path.join(extractDir, "agent_state"); + let vault: Buffer | null = null; + try { + vault = await fs.readFile(agentVaultPath(this.root, projectId, agentId)); + } catch { + // No vault means nothing to preserve. + } + if (vault !== null) { + await fs.writeFile(path.join(incomingState, VAULT_BASENAME), vault, { mode: 0o600 }); + } + const trash = path.join(staging, "replaced-agent_state"); + await fs.rename(stateDir, trash); + try { + await fs.rename(incomingState, stateDir); + } catch (err) { + await fs.rename(trash, stateDir); // Rollback: restore the old directory if swapping in the new one fails. + throw err; + } + return { version: incoming }; + } finally { + await fs.rm(staging, { recursive: true, force: true }); + } + } +} diff --git a/packages/server/src/services/trace-service.ts b/packages/server/src/services/trace-service.ts new file mode 100644 index 0000000..b9f2691 --- /dev/null +++ b/packages/server/src/services/trace-service.ts @@ -0,0 +1,683 @@ +/** + * Trace service. + * + * History messages: all of the Session's index files concatenated in order + * (readTraceTolerant, tolerating a truncated last line), containing only the + * complete messages and events that were actually written to Trace (naturally + * excluding partial_*); in-flight increments are continued by SSE. + * Performance analysis is derived from a single Trace file: nearest-neighbor + * pairing of request_begin/end, tool call duration pairing, reconnect / compaction + * counts, and Token trend. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { agentsDir, readTraceTolerant, tracesDir } from "@prismshadow/penguin-core"; +import type { OmniMessage } from "@prismshadow/penguin-core"; +import type { + AgentTracesResponse, + RequestSpan, + ToolCallSpan, + TraceAnalysisResponse, + TraceEventsResponse, + TraceFileInfo, + TraceModelSegment, + TraceTaskStats, + TraceToolSpan, + UsageTrendPointInTrace, +} from "../api/types.js"; +import { HttpError } from "../http/errors.js"; + +const TRACE_FILE_RE = /^(.+)_(\d{3})\.jsonl$/; + +/** Recursion depth cap for sub-session expansion (run_subagent depth is already constrained by the SDK; this is just a defensive backstop against cycles). */ +const MAX_SUBAGENT_DEPTH = 4; + +interface LocatedFile { + path: string; + date: string; + index: number; +} + +/** + * A **direct sub-session pointer** (the `subagent` event in the parent Trace) -> + * the sub-session's Session id. The pointer only + * records the Session id; the sub-session's Agent is located within the Project by + * its Trace file. + */ +function subagentPointer(msg: OmniMessage): string | null { + if (msg.type !== "event_msg") return null; + const p = msg.payload as { type?: string; session_id?: unknown }; + if (p.type !== "subagent" || typeof p.session_id !== "string" || p.session_id === "") { + return null; + } + return p.session_id; +} + +async function listDirs(dir: string): Promise { + try { + const entries = await fs.readdir(dir, { withFileTypes: true }); + return entries.filter((e) => e.isDirectory()).map((e) => e.name); + } catch { + return []; + } +} + +async function listFiles(dir: string): Promise { + try { + const entries = await fs.readdir(dir, { withFileTypes: true }); + return entries.filter((e) => e.isFile()).map((e) => e.name); + } catch { + return []; + } +} + +export class TraceService { + constructor(private readonly root: string) {} + + /** All of this Session's Trace files (sorted by index ascending). */ + private async locateAll( + projectId: string, + agentId: string, + sessionId: string, + ): Promise { + const dir = tracesDir(this.root, projectId, agentId); + const out: LocatedFile[] = []; + for (const dateDir of await listDirs(dir)) { + for (const file of await listFiles(path.join(dir, dateDir))) { + const match = TRACE_FILE_RE.exec(file); + if (!match || match[1] !== sessionId) continue; + out.push({ path: path.join(dir, dateDir, file), date: dateDir, index: Number(match[2]) }); + } + } + return out.sort((a, b) => a.index - b.index); + } + + /** Deletes all of this Session's Trace files (called when the Session is deleted). */ + async deleteSessionTraces(projectId: string, agentId: string, sessionId: string): Promise { + const files = await this.locateAll(projectId, agentId, sessionId); + for (const file of files) { + await fs.rm(file.path, { force: true }); + } + } + + /** + * History messages: all index files concatenated in order, with sub-sessions + * **expanded in place**. + * + * The parent Trace only records a `subagent` pointer event at the spawn point + * (recording just the child Session id; the content lives in the child + * Session's own Trace). Here the pointer is used to locate the child Trace + * within the Project, read it recursively, and splice the child messages — + * tagged with an origin chain — back in at the pointer's position, so that when + * the session is reopened, the frontend can re-attach the sub-session to the + * run_subagent tool card via origin (the child Trace's first `session_meta`, + * once given an origin, takes the same shape as what's forwarded over the live + * stream). When expansion succeeds, the pointer event itself is no longer + * emitted; when the child Trace is missing (deleted), the pointer event is kept + * so API consumers can still know it existed. + */ + async readMessages( + projectId: string, + agentId: string, + sessionId: string, + ): Promise { + return this.readMessagesExpanded(projectId, agentId, sessionId, { + index: null, + ancestry: new Set([sessionId]), + depth: 0, + }); + } + + /** + * A Project-wide session location index (sessionId -> agentId): built by + * scanning every Agent's traces directory. Built lazily the first time a + * subagent pointer is encountered, then reused across the whole readMessages + * call — rescanning per pointer would blow up into tens of thousands of readdir + * calls under multiple sub-sessions plus recursive expansion. + */ + private async buildSessionIndex(projectId: string): Promise> { + const index = new Map(); + for (const agentId of await listDirs(agentsDir(this.root, projectId))) { + const dir = tracesDir(this.root, projectId, agentId); + for (const dateDir of await listDirs(dir)) { + for (const file of await listFiles(path.join(dir, dateDir))) { + const match = TRACE_FILE_RE.exec(file); + if (match && !index.has(match[1]!)) index.set(match[1]!, agentId); + } + } + } + return index; + } + + private async readMessagesExpanded( + projectId: string, + agentId: string, + sessionId: string, + ctx: { index: Map | null; ancestry: Set; depth: number }, + ): Promise { + const files = await this.locateAll(projectId, agentId, sessionId); + const out: OmniMessage[] = []; + for (const file of files) { + for (const msg of await readTraceTolerant(file.path)) { + // The depth cap guards against runaway recursion; ancestry guards against a + // cyclic pointer (a tampered Trace pointing to itself/an ancestor is not expanded). + const childSid = ctx.depth < MAX_SUBAGENT_DEPTH ? subagentPointer(msg) : null; + if (!childSid || ctx.ancestry.has(childSid)) { + out.push(msg); + continue; + } + ctx.index ??= await this.buildSessionIndex(projectId); + const childAgent = ctx.index.get(childSid); + let nested: OmniMessage[] = []; + if (childAgent) { + ctx.ancestry.add(childSid); + nested = await this.readMessagesExpanded(projectId, childAgent, childSid, { + ...ctx, + depth: ctx.depth + 1, + }); + ctx.ancestry.delete(childSid); + } + // Child Trace missing (deleted): keep the pointer event, since the sub-session's content can't be recovered. + if (nested.length === 0) { + out.push(msg); + continue; + } + for (const m of nested) out.push({ ...m, origin: [childSid, ...(m.origin ?? [])] }); + } + } + return out; + } + + /** List of Trace files (index / date / size / mtime). */ + async listTraceFiles( + projectId: string, + agentId: string, + sessionId: string, + ): Promise { + const files = await this.locateAll(projectId, agentId, sessionId); + const out: TraceFileInfo[] = []; + for (const file of files) { + const stat = await fs.stat(file.path); + out.push({ + index: file.index, + date: file.date, + sizeBytes: stat.size, + mtime: stat.mtime.toISOString(), + }); + } + return out; + } + + /** Reads events from the Trace file at the given index, paginated by line (for loading large files in pages). */ + async readEvents( + projectId: string, + agentId: string, + sessionId: string, + index: number, + offset: number, + limit: number, + ): Promise { + const messages = await this.readFileByIndex(projectId, agentId, sessionId, index); + return { + events: messages.slice(offset, offset + limit), + offset, + limit, + total: messages.length, + }; + } + + /** Performance analysis: derived from a single Trace file. */ + async analyze( + projectId: string, + agentId: string, + sessionId: string, + index: number, + ): Promise { + const messages = await this.readFileByIndex(projectId, agentId, sessionId, index); + + const requests: RequestSpan[] = []; + let openRequest: RequestSpan | null = null; + const toolCalls: ToolCallSpan[] = []; + const openToolCalls = new Map(); + let reconnectCount = 0; + let compactionCount = 0; + const usageTrend: UsageTrendPointInTrace[] = []; + // Timeline (serial-duration estimation): Trace records completion times; model + // messages are produced + // serially (autoregressive decoding), so each segment's start = the previous + // event's time (the request's first segment = request_begin); a tool's + // approval/execution runs in parallel with model decoding, on its own lane; + // prevSerialTs is cleared after request_end, and the next request_begin + // restarts the count (which presumes all of the previous round's + // tool_call_output have already come back). + const modelSegments: TraceModelSegment[] = []; + const toolSpans: TraceToolSpan[] = []; + const openSpansById = new Map(); + let prevSerialTs: string | null = null; + // Task grouping: one user turn contains multiple Request rounds (the Agent + // loop sends another round each time it calls a tool); the turn ends once the + // model produces only text with no further tool call. Consecutive Requests are + // merged into one Task on this basis, and each Task gets its own independent + // timeline — different Tasks can be far apart in time (the user is thinking or + // has stepped away), and sharing one timeline would leave large gaps. + // Compaction forms its own turn: both compaction_begin/compaction_end break a + // continuation, so the compaction request becomes its own Task, and the + // request that resumes after compaction starts yet another Task. + let taskIndex = -1; + let continuation = false; // The previous round's Request called a tool -> the next request_begin continues the same Task + let sawToolCallThisRequest = false; + // Compaction interval (compaction_begin..compaction_end): the compaction + // request's request_begin/request_end and token_usage all fall inside it (see + // core context-engine's summarize flow), which is used to exclude the + // compaction request entirely from TPS — matching the same convention as + // compactionActive in the Chat page's task-stats. + let compactionActive = false; + // Token / duration totals per Task (computed server-side over the whole file: + // frontend events are fetched in pages, so summing them there would be + // mismatched). + const taskStats = new Map(); + const ensureTask = (ti: number): TraceTaskStats => { + let t = taskStats.get(ti); + if (t === undefined) { + t = { + taskIndex: ti, + messageFrom: -1, + messageTo: -1, + startTs: "", + endTs: "", + tokens: { cacheRead: 0, cacheWrite: 0, output: 0 }, + llmMs: 0, + }; + taskStats.set(ti, t); + } + return t; + }; + + /** + * Which turn each message belongs to: **decided definitively in one + * sequential pass**, not left for the frontend to guess by timestamp. + * + * Timestamp boundaries can't be pulled apart — the same millisecond can + * contain "the previous turn's last reply, compaction_begin, the compaction + * prompt, and the next turn's request_begin" all at once, so assigning by + * time would inevitably misfile this turn's reply into the next turn. + * + * Rule (a turn = one user turn; `request_end` is the end of some Request + * within a turn): + * - The **starting marker** of a new turn: the main session's user Prompt + * (outside compaction), or compaction_begin (compaction forms its own + * turn). Messages after the marker and before that turn's first + * `request_begin` (subsequent images from a multi-image send, the + * compaction prompt) are always held pending, waiting for + * `request_begin` to settle the new taskIndex before the whole span is + * assigned at once — they belong to the **new** turn, not the tail of + * the previous one. + * - Other messages belong to the current taskIndex: tool output and + * approval decisions arriving after request_end still belong to this + * turn (they're the results of tools this turn's Request initiated). + */ + const msgTask: number[] = new Array(messages.length).fill(-1); + /** The pending new turn's starting point (message index); settled once request_begin determines the taskIndex. */ + let pendingFrom: number | null = null; + + for (let mi = 0; mi < messages.length; mi++) { + const msg = messages[mi]!; + const p = msg.payload as Record & { type?: string }; + // The timeline only looks at the main session (a Trace itself never contains origin messages; this is a defensive skip). + const hasOrigin = msg.origin !== undefined && msg.origin.length > 0; + + // Starting marker of a new turn: the main session's user Prompt (outside + // compaction) -> a new user turn; compaction_begin -> a compaction turn + // (compaction forms its own turn). A single send can be "text + multiple + // images" = multiple messages; only the **first** one counts (once + // pendingFrom is set, it's not changed again), otherwise the turn's start + // would shift to the last image. + const startsUserTurn = + !hasOrigin && + !compactionActive && + msg.type === "model_msg" && + ((p.type === "text" && p.role === "user") || p.type === "image_url"); + const startsCompactionTurn = + !hasOrigin && msg.type === "event_msg" && p.type === "compaction_begin"; + if (startsUserTurn || startsCompactionTurn) { + if (pendingFrom === null) pendingFrom = mi; + // A user Prompt **always starts a new turn**: judging continuation solely + // by "did the previous turn call a tool" isn't enough — if the previous + // turn ended in timeout/malformed (given up after exhausting retries), + // retryable would leave continuation at true, and this new message would + // get merged into that failed turn, smearing the two turns' messages / + // Tokens / TPS / duration together. + if (startsUserTurn) continuation = false; + } + // A main-session message that isn't pending belongs to the current turn immediately (taskIndex < 0 = before the first request_begin, e.g. session_meta). + if (!hasOrigin && pendingFrom === null) msgTask[mi] = taskIndex; + + if (msg.type === "event_msg") { + if (p.type === "request_begin") { + if (!hasOrigin) { + prevSerialTs = msg.timestamp; + if (!continuation) taskIndex++; // Not a continuation -> a new Task + sawToolCallThisRequest = false; + } + // Settle taskIndex before opening the span: the span belongs directly to + // the current Task. Nearest-neighbor pairing: if the previous begin was + // never closed (process exited mid-run), the span is left open. + openRequest = { beginTs: msg.timestamp, taskIndex }; + if (compactionActive) openRequest.compaction = true; + requests.push(openRequest); + if (!hasOrigin) { + const t = ensureTask(taskIndex); + if (compactionActive) t.compaction = true; // This turn is a compaction turn + // This turn's duration starts at the first request_begin. It doesn't + // use the timestamp of the user Prompt / compaction summary or other + // user text: `` is created during compaction but only + // written to disk on the next run, so resuming the next day would + // stretch the first turn out by a whole day for no reason; the Prompt + // to request-dispatch gap is only ever milliseconds anyway. + if (t.startTs === "") t.startTs = msg.timestamp; + // The new turn's taskIndex is only settled here: the pending span + // (user Prompt / multiple images / compaction prompt) is assigned in + // full to **this** turn — they're the start of the new turn, not the + // tail of the previous one. + if (pendingFrom !== null) { + for (let k = pendingFrom; k < mi; k++) { + if (messages[k]!.origin === undefined) msgTask[k] = taskIndex; + } + pendingFrom = null; + } + msgTask[mi] = taskIndex; + } + } else if (p.type === "approval_decision") { + if (!hasOrigin && typeof p.tool_call_id === "string") { + const span = openSpansById.get(p.tool_call_id); + if (span && span.approvalTs === undefined) { + span.approvalTs = msg.timestamp; + if (typeof p.decision === "string") span.decision = p.decision; + // Approval wait time is subtracted out of the LLM generation + // duration: core does `await approve(tc)` inside the streaming loop, + // so the entire manual wait sits between request_begin and + // request_end (see RequestSpan.approvalWaitMs). Without subtracting + // it, "5s generation + 55s approval wait" would show 100 tok/s as 8 tok/s. + if (openRequest) { + const wait = Date.parse(msg.timestamp) - Date.parse(span.callTs); + if (Number.isFinite(wait) && wait > 0) { + openRequest.approvalWaitMs = (openRequest.approvalWaitMs ?? 0) + wait; + } + } + } + } + } else if (p.type === "request_end") { + const status = typeof p.status === "string" ? p.status : undefined; + // timeout/malformed is automatically reconnected by core within the same + // run (context-engine's retry loop); the resent Request still belongs to + // **the same user turn**: it must continue the turn, otherwise a single + // timeout would split that turn's Tokens/duration/TPS across two Tasks. + const retryable = status === "timeout" || status === "malformed"; + if (!hasOrigin) { + prevSerialTs = null; + continuation = sawToolCallThisRequest || retryable; + } + if (retryable) reconnectCount++; + if (openRequest) { + openRequest.endTs = msg.timestamp; + const dur = Date.parse(msg.timestamp) - Date.parse(openRequest.beginTs); + if (Number.isFinite(dur)) { + openRequest.durationMs = dur; + openRequest.activeMs = Math.max(0, dur - (openRequest.approvalWaitMs ?? 0)); + } + if (status !== undefined) openRequest.status = status; + // TPS denominator: accumulated per the turn a Request belongs to. A + // compaction request counts too — it belongs to **its own compaction + // turn** (compaction forms its own turn), so it neither pollutes a + // user turn's TPS, nor does the compaction turn fail to report its own + // generation speed accurately. A failed retry's duration is counted as + // well — it belongs to the same turn as the retry that eventually + // succeeded, and "how long this turn took to produce these tokens" + // should include the retries by definition. + if (!hasOrigin && openRequest.activeMs !== undefined) { + ensureTask(openRequest.taskIndex).llmMs += openRequest.activeMs; + } + openRequest = null; + } + } else if (p.type === "compaction_begin") { + compactionCount++; + // Compaction forms its own turn: otherwise, if the previous turn called + // a tool, continuation would still be true and the compaction request + // would get merged into the previous Task. + if (!hasOrigin) { + continuation = false; + compactionActive = true; + } + } else if (p.type === "compaction_end") { + // Both ends of compaction break a continuation. This closing one can't + // be skipped: if the compaction request itself exhausts its retries and + // ends in timeout, the retryable check above would mark it as "continued", + // and without clearing it here, the next user turn after compaction + // would get merged into this compaction Task. + if (!hasOrigin) { + continuation = false; + compactionActive = false; + } + } else if (p.type === "token_usage") { + const request = p.request as + | { total?: number; cache_read?: number; cache_write?: number; output?: number } + | undefined; + const session = p.session as { total?: number } | undefined; + usageTrend.push({ + ts: msg.timestamp, + requestTotal: request?.total ?? 0, + sessionTotal: session?.total ?? 0, + }); + if (!hasOrigin) { + const t = ensureTask(taskIndex); + // Cumulative usage for this turn (a running total): those tokens were + // actually paid for, so the cost can't be dropped. `tokens.output` also + // doubles as the numerator for output TPS — compaction's output + // belongs to **its own compaction turn** (compaction forms its own + // turn), so a user turn's TPS isn't polluted by it, while the + // compaction turn can still accurately report "how fast the summary + // was generated". + t.tokens.cacheRead += request?.cache_read ?? 0; + t.tokens.cacheWrite += request?.cache_write ?? 0; + t.tokens.output += request?.output ?? 0; + if (!compactionActive) { + // The context snapshot only takes non-compaction Requests: tokens + // consumed by compaction aren't the post-compaction context + // footprint. A later write overwrites an earlier one -> this + // naturally leaves behind the snapshot of the Task's **last** + // non-compaction Request = the context footprint at the end of this + // turn. Accumulating would be wrong: each Request's input carries + // the entire history afresh (see TraceTaskStats). + t.context = { + cacheRead: request?.cache_read ?? 0, + cacheWrite: request?.cache_write ?? 0, + output: request?.output ?? 0, + }; + } + } + } + continue; + } + if (msg.type !== "model_msg") continue; + // Model serial segments: assistant-side thinking/text/tool_call (a user input sent instantaneously occupies no segment). + if ( + !hasOrigin && + prevSerialTs !== null && + (p.type === "thinking" || + p.type === "tool_call" || + (p.type === "text" && p.role === "assistant")) + ) { + const segment: TraceModelSegment = { + kind: p.type === "thinking" ? "thinking" : p.type === "tool_call" ? "tool_call" : "text", + startTs: prevSerialTs, + endTs: msg.timestamp, + taskIndex, + }; + if (p.type === "tool_call" && typeof p.tool_call_id === "string") { + segment.toolCallId = p.tool_call_id; + if (typeof p.name === "string") segment.name = p.name; + } + modelSegments.push(segment); + prevSerialTs = msg.timestamp; + } + if (p.type === "tool_call" && typeof p.tool_call_id === "string") { + if (!hasOrigin) sawToolCallThisRequest = true; // This turn called a tool -> the next turn continues the same Task + const callStop = typeof p.stop_reason === "string" ? p.stop_reason : undefined; + const span: ToolCallSpan = { + toolCallId: p.tool_call_id, + name: typeof p.name === "string" ? p.name : "", + startTs: msg.timestamp, + }; + if (callStop !== undefined) span.stopReason = callStop; + openToolCalls.set(p.tool_call_id, span); + toolCalls.push(span); + // The timeline lane only accepts calls that "will actually be executed": + // an interrupt-compensation tool_call (stop_reason other than completed) + // never gets an approval/output, and putting it on a lane would render as + // a phantom "executing" state spanning the whole timeline — so it's + // skipped outright. + if (!hasOrigin && (callStop === undefined || callStop === "completed")) { + const timeline: TraceToolSpan = { + toolCallId: p.tool_call_id, + name: typeof p.name === "string" ? p.name : "", + callTs: msg.timestamp, + taskIndex, + }; + openSpansById.set(p.tool_call_id, timeline); + toolSpans.push(timeline); + } + } else if (p.type === "tool_call_output" && typeof p.tool_call_id === "string") { + const span = openToolCalls.get(p.tool_call_id); + if (span && span.endTs === undefined) { + span.endTs = msg.timestamp; + const dur = Date.parse(msg.timestamp) - Date.parse(span.startTs); + if (Number.isFinite(dur)) span.durationMs = dur; + if (typeof p.stop_reason === "string") span.stopReason = p.stop_reason; + } + const timeline = openSpansById.get(p.tool_call_id); + if (timeline && timeline.outputTs === undefined) { + timeline.outputTs = msg.timestamp; + if (typeof p.stop_reason === "string") timeline.stopReason = p.stop_reason; + } + } + } + + // A pending span that never got a request_begin (interrupted right after the + // user sent it / the process exited): it's a turn that never got to run, + // and forms its own turn — reattaching it to the previous turn would smear + // two separate user sends together. + if (pendingFrom !== null) { + taskIndex++; + for (let k = pendingFrom; k < messages.length; k++) { + if (messages[k]!.origin === undefined) msgTask[k] = taskIndex; + } + ensureTask(taskIndex); + } + + // Each turn's message index range and end-of-turn time are always derived from + // the per-message assignment done above (same source, so they never disagree + // with each other). Messages before the first request_begin (session_meta) + // have taskIndex -1 and are assigned to the first turn, otherwise they'd have + // nowhere to sit on the page. The turn duration's **starting point** isn't + // decided here — it was already settled at request_begin (duration only looks + // at LLM requests; timestamps of the user Prompt / compaction summary or other + // user text don't participate, see TraceTaskStats.startTs). + const firstTask = [...taskStats.keys()].sort((a, b) => a - b)[0]; + for (let k = 0; k < messages.length; k++) { + let ti = msgTask[k]!; + if (ti < 0) { + if (firstTask === undefined) continue; + ti = firstTask; + msgTask[k] = ti; + } + const t = ensureTask(ti); + if (t.messageFrom < 0 || k < t.messageFrom) t.messageFrom = k; + if (k > t.messageTo) t.messageTo = k; + // session_meta is only **listed** in the first turn, and doesn't count + // toward the end-of-turn time: it's metadata written when the session was + // created, and its timestamp has nothing to do with this turn (it also gets + // rewritten verbatim at the start of a new file after compaction splits the file). + if (messages[k]!.type === "session_meta") continue; + const ts = messages[k]!.timestamp; + if (t.endTs === "" || ts > t.endTs) t.endTs = ts; + } + + const tasks = [...taskStats.values()].sort((a, b) => a.taskIndex - b.taskIndex); + // Total elapsed time = **the sum of each turn's duration**, matching exactly + // the scope shown per-turn below (**including compaction turns** — their wall + // clock time genuinely elapsed, each turn's card has its own duration, and the + // overall total is their sum, so the numbers must add up). It is not "last + // message timestamp minus first message timestamp": that would be the whole + // file's wall-clock span, counting in the gaps **between** turns (the user + // thinking, stepping out for coffee, coming back the next day) — none of which + // is time the Agent spent working. A degenerate turn with no Request has an + // empty startTs and counts as 0. + // Note this uses a different convention from the Session's cumulative elapsed + // time on the Chat page: that one only accumulates user turns (compaction + // after a turn ends doesn't count toward the turn). + const elapsedMs = tasks.reduce((sum, t) => { + const span = Date.parse(t.endTs) - Date.parse(t.startTs); + return sum + (Number.isFinite(span) ? Math.max(0, span) : 0); + }, 0); + return { + elapsedMs, + requests, + tasks, + toolCalls, + modelSegments, + toolSpans, + reconnectCount, + compactionCount, + usageTrend, + }; + } + + /** Level-by-level browsing (newest first): Agent -> date -> Session -> Trace files. */ + async agentTraces(projectId: string, agentId: string): Promise { + const dir = tracesDir(this.root, projectId, agentId); + const dates = (await listDirs(dir)).sort().reverse(); + const out: AgentTracesResponse = { dates: [] }; + for (const date of dates) { + const bySession = new Map(); + for (const file of await listFiles(path.join(dir, date))) { + const match = TRACE_FILE_RE.exec(file); + if (!match) continue; + const sessionId = match[1]!; + const stat = await fs.stat(path.join(dir, date, file)); + const files = bySession.get(sessionId) ?? []; + files.push({ index: Number(match[2]), sizeBytes: stat.size }); + bySession.set(sessionId, files); + } + if (bySession.size === 0) continue; + out.dates.push({ + date, + // session_id embeds a timestamp, so reverse lexicographic order is reverse chronological order. + sessions: [...bySession.entries()] + .sort((a, b) => b[0].localeCompare(a[0])) + .map(([sessionId, files]) => ({ + sessionId, + files: files.sort((a, b) => a.index - b.index), + })), + }); + } + return out; + } + + private async readFileByIndex( + projectId: string, + agentId: string, + sessionId: string, + index: number, + ): Promise { + const files = await this.locateAll(projectId, agentId, sessionId); + const file = files.find((f) => f.index === index); + if (!file) { + throw new HttpError( + 404, + "trace_not_found", + `该 Session 没有 index 为 ${index} 的 Trace 文件。`, + ); + } + return readTraceTolerant(file.path); + } +} diff --git a/packages/server/src/services/usage-service.ts b/packages/server/src/services/usage-service.ts new file mode 100644 index 0000000..18421d3 --- /dev/null +++ b/packages/server/src/services/usage-service.ts @@ -0,0 +1,269 @@ +/** + * Usage statistics query. + * + * Cost is **computed in real time**: usage_records only stores Tokens (pricing may + * be added later), so at query time each Model's cost is converted using the + * current Project's configured pricing — the repo returns raw Token totals broken + * down by `(provider, model_id)` paired reference, and this service looks up each + * reference's price once and folds it into cost / hasUncosted (if a Model has no + * pricing, its consumption is excluded from cost and hasUncosted is flagged). + * Summary cards (today / last 7 days / cumulative), grouped aggregation (date / + * agent / model / session, with the session dimension supporting agentId drill-down + * filtering), and a 30-day trend. + * Server-side error statistics (error_records) ride along on the same response: + * the statistics center fetches everything in one request, and filters are + * naturally shared; unattributed errors (login failures, process crashes, and other + * errors with no Project context) are visible only to admins, see the ErrorsRepo + * file header. + */ +import type { + UsageBucket, + UsageErrors, + UsageGroupBy, + UsageGroupRow, + UsageResponse, +} from "../api/types.js"; +import type { ErrorFilter, ErrorsRepo } from "../db/repos/errors.js"; +import type { + UsageRepo, + UsageModelSums, + UsageGroupModelSums, + UsageFilter, +} from "../db/repos/usage.js"; +import { formatLocalDate, localDateMinusDays } from "../internal/dates.js"; + +/** Number of most-recent entries kept in the error detail table. */ +const ERROR_RECENT_N = 20; + +/** The three pricing buckets (usd_per_mtok convention), returned by the pricing lookup callback. */ +export interface PricingRates { + cacheRead: number; + cacheWrite: number; + output: number; +} + +export type PricingLookup = ( + projectId: string, + provider: string, + modelId: string, +) => Promise; + +export interface UsageQuery { + from?: string; + to?: string; + groupBy: UsageGroupBy; + /** Top-level filter: view by Agent (also used for groupBy=session drill-down). */ + agentId?: string; + /** Top-level filter: view by Model (paired with modelId; the dropdown always sends them as a pair). */ + provider?: string; + modelId?: string; + /** Whether to include unattributed errors: admin only (the route passes user.isAdmin), defaults to false. */ + includeGlobalErrors?: boolean; +} + +/** Cost formula: sum of the three buckets, in USD per million Tokens. */ +function costOf(sums: UsageModelSums, rates: PricingRates): number { + return ( + (sums.cacheRead * rates.cacheRead + + sums.cacheWrite * rates.cacheWrite + + sums.output * rates.output) / + 1e6 + ); +} + +/** In-process Map key for a paired reference (\0-separated, the same style as session-manager's agentKey; never persisted). */ +function refKey(provider: string, modelId: string): string { + return `${provider}\0${modelId}`; +} + +export class UsageService { + constructor( + private readonly usage: UsageRepo, + private readonly errors: ErrorsRepo, + private readonly lookupPricing: PricingLookup, + private readonly now: () => Date = () => new Date(), + ) {} + + async query(projectId: string, q: UsageQuery): Promise { + const today = formatLocalDate(this.now()); + // Top-level filter: agent + model (the cost center switches views by agent/model; the model filter is always sent as a pair). + const base: UsageFilter = {}; + if (q.agentId !== undefined) base.agentId = q.agentId; + if (q.provider !== undefined) base.provider = q.provider; + if (q.modelId !== undefined) base.modelId = q.modelId; + + const win = (from?: string, to?: string): UsageFilter => ({ + ...base, + ...(from !== undefined ? { from } : {}), + ...(to !== undefined ? { to } : {}), + }); + + const todayRows = this.usage.bucketByModel(projectId, win(today, today)); + const last7dRows = this.usage.bucketByModel(projectId, win(localDateMinusDays(this.now(), 6))); + const totalRows = this.usage.bucketByModel(projectId, win(q.from, q.to)); + const groupRows = this.usage.groupsByModel(projectId, q.groupBy, win(q.from, q.to)); + // Fixed 30-day window; affected by the agent/model filter. + const trendFrom = localDateMinusDays(this.now(), 29); + const trendRows = this.usage.groupsByModel(projectId, "date", win(trendFrom)); + // Agent call-count chart: not affected by the agent filter (shows all agents), but still affected by the date + model filter. + const agentRows = this.usage.groupsByModel(projectId, "agent", { + ...(q.provider !== undefined ? { provider: q.provider } : {}), + ...(q.modelId !== undefined ? { modelId: q.modelId } : {}), + ...(q.from !== undefined ? { from: q.from } : {}), + ...(q.to !== undefined ? { to: q.to } : {}), + }); + // Model success-rate chart: not affected by the model filter (shows all models), but still affected by the date + agent filter. + const statusRows = this.usage.statusByModel(projectId, { + ...(q.agentId !== undefined ? { agentId: q.agentId } : {}), + ...(q.from !== undefined ? { from: q.from } : {}), + ...(q.to !== undefined ? { to: q.to } : {}), + }); + // Error statistics: likewise not affected by the model filter (HTTP / process errors have no Model dimension), but still affected by the date + agent filter. + const errorFilter: ErrorFilter = { + ...(q.agentId !== undefined ? { agentId: q.agentId } : {}), + ...(q.from !== undefined ? { from: q.from } : {}), + ...(q.to !== undefined ? { to: q.to } : {}), + // Unattributed errors are visible only to admins (regular members only see errors within their own Project, see the ErrorsRepo file header). + ...(q.includeGlobalErrors === true ? { includeGlobal: true } : {}), + }; + + // Each paired reference that occurs is looked up for its current price only once. + const rates = new Map(); + const allRefs = new Map(); + for (const r of [...todayRows, ...last7dRows, ...totalRows, ...groupRows, ...trendRows]) { + allRefs.set(refKey(r.provider, r.modelId), { provider: r.provider, modelId: r.modelId }); + } + for (const [key, ref] of allRefs) { + rates.set(key, await this.lookupPricing(projectId, ref.provider, ref.modelId)); + } + + const byAgentMap = new Map(); + for (const r of agentRows) { + const acc = byAgentMap.get(r.key) ?? { requests: 0, total: 0 }; + acc.requests += r.requests; + acc.total += r.total; + byAgentMap.set(r.key, acc); + } + + return { + summary: { + today: this.foldBucket(todayRows, rates), + last7d: this.foldBucket(last7dRows, rates), + total: this.foldBucket(totalRows, rates), + }, + groupBy: q.groupBy, + groups: this.foldGroups(groupRows, rates, q.groupBy), + trend: this.foldTrend(trendRows, rates), + byAgent: [...byAgentMap.entries()] + .map(([agentId, v]) => ({ agentId, requests: v.requests, total: v.total })) + .sort((a, b) => b.requests - a.requests), + success: statusRows.sort((a, b) => b.total - a.total), + errors: this.foldErrors(projectId, errorFilter), + agentIds: this.usage.distinctAgentIds(projectId), + models: this.usage.distinctModels(projectId), + }; + } + + /** Error statistics: summary info (total / unexpected / most common error code) + the last N entries, all filtered by the selected range. */ + private foldErrors(projectId: string, f: ErrorFilter): UsageErrors { + const { total, unexpected } = this.errors.summary(projectId, f); + return { + total, + unexpected, + topCode: this.errors.topCode(projectId, f), + recent: this.errors.recent(projectId, f, ERROR_RECENT_N), + }; + } + + private foldBucket( + rows: UsageModelSums[], + rates: Map, + ): UsageBucket { + let total = 0; + let requests = 0; + let cost: number | null = null; + let hasUncosted = false; + for (const r of rows) { + total += r.total; + requests += r.requests; + const rate = rates.get(refKey(r.provider, r.modelId)); + if (rate) cost = (cost ?? 0) + costOf(r, rate); + else hasUncosted = true; + } + return { total, requests, cost, hasUncosted }; + } + + private foldGroups( + rows: UsageGroupModelSums[], + rates: Map, + groupBy: UsageGroupBy, + ): UsageGroupRow[] { + // The model dimension folds by paired reference (a shared model_id name across providers is split into separate rows); other dimensions fold by their group key. + const keyOf = (r: UsageGroupModelSums): string => + groupBy === "model" ? refKey(r.provider, r.key) : r.key; + const byKey = new Map(); + for (const r of rows) { + const acc = byKey.get(keyOf(r)) ?? { + key: r.key, + ...(groupBy === "model" ? { provider: r.provider } : {}), + cacheRead: 0, + cacheWrite: 0, + output: 0, + total: 0, + requests: 0, + cost: null as number | null, + hasUncosted: false, + }; + acc.cacheRead += r.cacheRead; + acc.cacheWrite += r.cacheWrite; + acc.output += r.output; + acc.total += r.total; + acc.requests += r.requests; + const rate = rates.get(refKey(r.provider, r.modelId)); + if (rate) acc.cost = (acc.cost ?? 0) + costOf(r, rate); + else acc.hasUncosted = true; + byKey.set(keyOf(r), acc); + } + const out = [...byKey.values()]; + // The date dimension sorts by key descending (most recent first); other dimensions sort by total Token count descending. + if (groupBy === "date") out.sort((a, b) => b.key.localeCompare(a.key)); + else out.sort((a, b) => b.total - a.total); + return out; + } + + private foldTrend( + rows: UsageGroupModelSums[], + rates: Map, + ): UsageResponse["trend"] { + const byDate = new Map< + string, + { total: number; cacheRead: number; cacheWrite: number; output: number; cost: number | null } + >(); + for (const r of rows) { + const acc = byDate.get(r.key) ?? { + total: 0, + cacheRead: 0, + cacheWrite: 0, + output: 0, + cost: null, + }; + acc.total += r.total; + acc.cacheRead += r.cacheRead; + acc.cacheWrite += r.cacheWrite; + acc.output += r.output; + const rate = rates.get(refKey(r.provider, r.modelId)); + if (rate) acc.cost = (acc.cost ?? 0) + costOf(r, rate); + byDate.set(r.key, acc); + } + return [...byDate.entries()] + .sort((a, b) => a[0].localeCompare(b[0])) + .map(([date, v]) => ({ + date, + total: v.total, + cacheRead: v.cacheRead, + cacheWrite: v.cacheWrite, + output: v.output, + cost: v.cost, + })); + } +} diff --git a/packages/server/src/services/workspace-files-service.ts b/packages/server/src/services/workspace-files-service.ts new file mode 100644 index 0000000..32be382 --- /dev/null +++ b/packages/server/src/services/workspace-files-service.ts @@ -0,0 +1,283 @@ +/** + * Workspace file browsing: list directory / read + * file (preview & download) / write file (upload). Security: a relative path, once + * resolved, must stay inside the Workspace — a logical prefix check plus a realpath + * check against the nearest existing ancestor (guards against `..` and symlink escapes). + */ +import fs from "node:fs/promises"; +import { constants as fsc } from "node:fs"; +import path from "node:path"; +import type { WorkspaceFilesResponse } from "../api/types.js"; +import { HttpError } from "../http/errors.js"; +import { badRequest } from "../http/validate.js"; + +/** Per-file read cap (a safety limit since preview/download reads the whole file into memory). */ +const MAX_READ_BYTES = 50 * 1024 * 1024; +/** Upload cap (stays within the 20MB request body limit even after base64 encoding). */ +export const MAX_UPLOAD_BYTES = 14 * 1024 * 1024; + +const CONTENT_TYPES: Record = { + ".html": "text/html; charset=utf-8", + ".htm": "text/html; charset=utf-8", + ".txt": "text/plain; charset=utf-8", + ".md": "text/markdown; charset=utf-8", + ".json": "application/json", + ".js": "text/javascript; charset=utf-8", + ".ts": "text/plain; charset=utf-8", + ".tsx": "text/plain; charset=utf-8", + ".py": "text/plain; charset=utf-8", + ".sh": "text/plain; charset=utf-8", + ".yaml": "text/plain; charset=utf-8", + ".yml": "text/plain; charset=utf-8", + ".toml": "text/plain; charset=utf-8", + ".css": "text/css; charset=utf-8", + ".csv": "text/plain; charset=utf-8", + ".log": "text/plain; charset=utf-8", + ".svg": "image/svg+xml", + ".png": "image/png", + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".gif": "image/gif", + ".webp": "image/webp", + ".pdf": "application/pdf", +}; + +export interface WorkspaceFileContent { + data: Buffer; + fileName: string; + contentType: string; + /** Types whose same-origin inline rendering would execute scripts (html/svg): inline preview must fall back to plain text. */ + scriptable: boolean; +} + +export class WorkspaceFilesService { + /** Canonical path (realpath) of the Workspace root; 404 if it doesn't exist. */ + private async realBase(workspace: string): Promise { + try { + return await fs.realpath(path.resolve(workspace)); + } catch { + throw new HttpError(404, "workspace_missing", "该 Session 的 Workspace 已不存在。"); + } + } + + /** + * Lexical containment check: whether target is inside base (including equal to + * base). Uses path.relative rather than prefix concatenation, so it works when + * base is the filesystem root ("/" concatenated with sep would produce a "//" + * prefix that no subpath could ever match); only a full ".." segment is + * compared, so a legitimate name like "..foo" isn't mistakenly rejected. + */ + private isInside(target: string, base: string): boolean { + const rel = path.relative(base, target); + return rel !== ".." && !rel.startsWith(`..${path.sep}`) && !path.isAbsolute(rel); + } + + /** Lexical prefix check (a relative path, once resolved, must still be inside the Workspace); returns the absolute target path. */ + private lexicalTarget(base: string, rel: string): string { + if (rel.includes("\0")) throw badRequest("path 非法。"); + const target = path.resolve(base, rel === "" ? "." : rel); + if (!this.isInside(target, base)) { + throw badRequest("path 必须位于 Workspace 内。"); + } + return target; + } + + private assertInside(real: string, realBase: string): void { + if (!this.isInside(real, realBase)) { + throw badRequest("path 必须位于 Workspace 内。"); + } + } + + /** + * Read-path resolution: realpath the entire path (following all symlinks to get + * a link-free canonical path), then check containment and **perform IO on the + * canonical path** — since the canonical path contains no symlink segments at + * all, this eliminates check-then-use TOCTOU escapes (an out-of-bounds symlink + * is already resolved and rejected at the realpath step). + */ + private async resolveRead(workspace: string, rel: string): Promise { + const realBase = await this.realBase(workspace); + const target = this.lexicalTarget(path.resolve(workspace), rel); + let canonical: string; + try { + canonical = await fs.realpath(target); + } catch (err) { + if ((err as NodeJS.ErrnoException).code === "ENOENT") { + throw new HttpError(404, "path_not_found", "文件不存在。"); + } + throw err; + } + this.assertInside(canonical, realBase); + return canonical; + } + + /** + * Write-path resolution: realpaths the parent directory (whose canonical path + * has no symlink segments) and checks containment, then appends the final + * segment as the file name. When the parent directory is missing, it is safely + * created (uploading a folder needs to preserve directory structure): first the + * nearest **existing** ancestor is found and its canonical path checked against + * the Workspace — this exposes it if a middle segment was preset as a symlink + * pointing outside; the missing segments are then created recursively beneath it + * (a brand-new directory can never be a symlink), followed by a second realpath + * check after creation. The actual write opens with O_NOFOLLOW (refusing to + * follow a symlink at the final segment), blocking the sandbox-escape pattern of + * "Agent presets a symlink -> an upload is used as leverage to overwrite a file + * outside the sandbox". Returns the canonical parent directory + file name. + */ + private async resolveWriteParent( + workspace: string, + rel: string, + ): Promise<{ dir: string; name: string }> { + const realBase = await this.realBase(workspace); + const target = this.lexicalTarget(path.resolve(workspace), rel); + const name = path.basename(target); + if (name === "" || name === "." || name === "..") throw badRequest("path 必须是文件路径。"); + const parent = path.dirname(target); + let canonicalParent: string; + try { + canonicalParent = await fs.realpath(parent); + } catch (err) { + if ((err as NodeJS.ErrnoException).code !== "ENOENT") throw err; + let probe = parent; + while (true) { + try { + this.assertInside(await fs.realpath(probe), realBase); + break; + } catch (probeErr) { + if ((probeErr as NodeJS.ErrnoException).code !== "ENOENT") throw probeErr; + const up = path.dirname(probe); + if (up === probe) throw badRequest("path 非法。"); + probe = up; + } + } + await fs.mkdir(parent, { recursive: true }); + canonicalParent = await fs.realpath(parent); + } + this.assertInside(canonicalParent, realBase); + return { dir: canonicalParent, name }; + } + + /** + * Batch existence check (a message's file card lists only files that actually + * exist): each item goes through the same containment resolution as reading + * (resolveRead); out-of-bounds, resolution failure, missing Workspace, or an + * irregular file are all treated as non-existent — the card scenario only asks + * "can this be opened", and throwing a 4xx would only add frontend branches while + * leaking containment details. Returns the deduplicated existing items in input order. + */ + async statExisting(workspace: string, rels: string[]): Promise { + const unique = [...new Set(rels)]; + const exists = await Promise.all( + unique.map(async (rel) => { + try { + const stat = await fs.stat(await this.resolveRead(workspace, rel)); + return stat.isFile(); + } catch { + return false; + } + }), + ); + return unique.filter((_, i) => exists[i]); + } + + /** List a directory: dirs come first, each group sorted by name; kind follows the symlink target (consistent with read behavior). */ + async list(workspace: string, rel: string): Promise { + const dir = await this.resolveRead(workspace, rel); + let dirents; + try { + dirents = await fs.readdir(dir, { withFileTypes: true }); + } catch (err) { + if ((err as NodeJS.ErrnoException).code === "ENOENT") { + throw new HttpError(404, "path_not_found", "目录不存在。"); + } + if ((err as NodeJS.ErrnoException).code === "ENOTDIR") { + throw badRequest("path 不是目录。"); + } + throw err; + } + const entries = await Promise.all( + dirents.map(async (d) => { + let sizeBytes = 0; + let mtime = ""; + // Dirent doesn't report the target type for a symlink, so stat (following the link) is used to determine dir/file. + let isDir = d.isDirectory(); + try { + const stat = await fs.stat(path.join(dir, d.name)); + sizeBytes = stat.size; + mtime = stat.mtime.toISOString(); + isDir = stat.isDirectory(); + } catch { + // A dangling symlink or similar: keep the entry, with size/time left at defaults. + } + return { + name: d.name, + kind: isDir ? ("dir" as const) : ("file" as const), + sizeBytes, + mtime, + }; + }), + ); + entries.sort((a, b) => + a.kind === b.kind ? a.name.localeCompare(b.name) : a.kind === "dir" ? -1 : 1, + ); + return { path: rel, entries }; + } + + /** Read a file (preview/download): IO on the canonical path (resolveRead has already eliminated symlink escapes). */ + async read(workspace: string, rel: string): Promise { + const file = await this.resolveRead(workspace, rel); + let stat; + try { + stat = await fs.stat(file); + } catch { + throw new HttpError(404, "path_not_found", "文件不存在。"); + } + if (stat.isDirectory()) throw badRequest("path 是目录。"); + if (stat.size > MAX_READ_BYTES) { + throw new HttpError(413, "file_too_large", "文件超过 50MB 读取上限。"); + } + const data = await fs.readFile(file); + const ext = path.extname(file).toLowerCase(); + return { + data, + fileName: path.basename(file), + contentType: CONTENT_TYPES[ext] ?? "application/octet-stream", + scriptable: ext === ".html" || ext === ".htm" || ext === ".svg", + }; + } + + /** + * Write a file (upload, overwriting a same-named one). If the parent directory + * is missing, it's automatically created under sandbox checks (preserving + * directory structure for folder uploads); the final segment is opened with + * O_NOFOLLOW, refusing to follow a symlink to write outside the Workspace + * (together with resolveWriteParent's canonical-parent check, this blocks + * sandbox escapes). + */ + async write(workspace: string, rel: string, data: Buffer): Promise { + if (rel === "" || rel.endsWith("/")) throw badRequest("path 必须是文件路径。"); + if (data.length > MAX_UPLOAD_BYTES) { + throw new HttpError(413, "file_too_large", "上传文件超过 14MB 上限。"); + } + const { dir, name } = await this.resolveWriteParent(workspace, rel); + const file = path.join(dir, name); + // O_NOFOLLOW: open reports ELOOP if the final segment is a symlink, refusing to use it as leverage to overwrite a file outside the sandbox. + const flags = fsc.O_WRONLY | fsc.O_CREAT | fsc.O_TRUNC | (fsc.O_NOFOLLOW ?? 0); + let handle; + try { + handle = await fs.open(file, flags, 0o644); + } catch (err) { + const code = (err as NodeJS.ErrnoException).code; + if (code === "ELOOP") throw badRequest("path 不能是符号链接。"); + if (code === "ENOENT") throw new HttpError(404, "path_not_found", "父目录不存在。"); + if (code === "EISDIR") throw badRequest("path 是目录。"); + throw err; + } + try { + await handle.writeFile(data); + } finally { + await handle.close(); + } + } +} diff --git a/packages/server/src/services/workspace-guard.ts b/packages/server/src/services/workspace-guard.ts new file mode 100644 index 0000000..daaa8fe --- /dev/null +++ b/packages/server/src/services/workspace-guard.ts @@ -0,0 +1,37 @@ +/** + * Workspace validation. + * + * When a user explicitly specifies a Workspace: after realpath normalization + * (resolving `..` and symlinks), it's required to be an **existing directory** + * (never auto-created). The directory location is not constrained to the + * Project directory — a Workspace can be any path on the server, with actual + * reachability governed by the file permissions of the OS account running the + * service. + * When no Workspace is specified, this module isn't involved (the SDK creates its + * own temporary directory). + */ +import fs from "node:fs/promises"; +import { HttpError } from "../http/errors.js"; + +/** + * Validates and returns the normalized (realpath) Workspace path. + * + * @throws 400 workspace_not_found: the path doesn't exist, isn't readable, or isn't a directory. + */ +export async function assertWorkspaceAllowed(args: { workspace: string }): Promise { + let ws: string; + try { + ws = await fs.realpath(args.workspace); + } catch { + throw new HttpError( + 400, + "workspace_not_found", + `Workspace 不存在或不可访问:${args.workspace}。请指定一个已存在的目录,或留空以使用临时目录。`, + ); + } + const stat = await fs.stat(ws); + if (!stat.isDirectory()) { + throw new HttpError(400, "workspace_not_found", `Workspace 不是目录:${args.workspace}。`); + } + return ws; +} diff --git a/packages/server/test/admin-users.test.ts b/packages/server/test/admin-users.test.ts new file mode 100644 index 0000000..bf671af --- /dev/null +++ b/packages/server/test/admin-users.test.ts @@ -0,0 +1,138 @@ +/** + * Admin users backend integration tests: permission boundary (non-admin 403), account + * creation validation and rollback, password reset (session invalidation), and user + * deletion (cascading deletion of owned Project). + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import type { AdminUsersResponse, MembersResponse } from "../src/api/types.js"; +import { apiClient, createTestApp, loginAdmin, loginUser, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +describe("admin 用户后台", () => { + let t: TestApp; + let admin: ReturnType; + + beforeEach(async () => { + t = await createTestApp(); + admin = apiClient(t.app, (await loginAdmin(t.app)).cookie); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("非管理员访问一律 403", async () => { + const { cookie } = await provisionUser(t.app, "norm"); + const api = apiClient(t.app, cookie); + expect((await api.get("/api/admin/users")).status).toBe(403); + expect( + (await api.post("/api/admin/users", { userId: "x_user", password: "password-123" })).status, + ).toBe(403); + expect( + (await api.post("/api/admin/users/norm/password", { password: "password-456" })).status, + ).toBe(403); + expect((await api.delete("/api/admin/users/norm")).status).toBe(403); + }); + + it("建号校验:非法用户名 / 过短密码 400;重复 409", async () => { + for (const bad of ["Bob", "1abc", "a", "-abc", "a-b", "a!b", "a".repeat(33)]) { + const res = await admin.post("/api/admin/users", { userId: bad, password: "password-123" }); + expect(res.status, `userId=${bad}`).toBe(400); + } + expect( + (await admin.post("/api/admin/users", { userId: "ok_user", password: "short" })).status, + ).toBe(400); + expect( + (await admin.post("/api/admin/users", { userId: "ok_user", password: "password-123" })) + .status, + ).toBe(201); + expect( + (await admin.post("/api/admin/users", { userId: "ok_user", password: "password-123" })) + .status, + ).toBe(409); + expect( + (await admin.post("/api/admin/users", { userId: "admin", password: "password-123" })).status, + ).toBe(409); + }); + + it("默认 Project id 被占用时建号失败并回滚用户", async () => { + // Occupy the frank-default_project directory (simulating CLI creation; no Web-side user can construct another's prefix). + await fs.mkdir(path.join(t.root, "frank-default_project"), { recursive: true }); + const res = await admin.post("/api/admin/users", { userId: "frank", password: "password-123" }); + expect(res.status).toBe(409); + // The user row has been rolled back: frank is absent from the list and cannot log in. + const list = (await (await admin.get("/api/admin/users")).json()) as AdminUsersResponse; + expect(list.users.map((u) => u.userId)).not.toContain("frank"); + }); + + it("用户列表:种子 admin 与新建用户,字段齐全", async () => { + await provisionUser(t.app, "kate"); + const list = (await (await admin.get("/api/admin/users")).json()) as AdminUsersResponse; + expect(list.users.map((u) => u.userId)).toEqual(["admin", "kate"]); + const a = list.users[0]!; + expect(a.isAdmin).toBe(true); + expect(a.passwordIsInitial).toBe(true); + expect(a.createdAt).toBeTruthy(); + expect(list.users[1]!.isAdmin).toBe(false); + }); + + it("重置密码:清空目标用户全部会话,新密码带初始标记", async () => { + const rex = await provisionUser(t.app, "rex"); + const rexApi = apiClient(t.app, rex.cookie); + expect((await rexApi.get("/api/me")).status).toBe(200); + + expect( + (await admin.post("/api/admin/users/ghost/password", { password: "password-456" })).status, + ).toBe(404); + expect((await admin.post("/api/admin/users/rex/password", { password: "short" })).status).toBe( + 400, + ); + expect( + (await admin.post("/api/admin/users/rex/password", { password: "password-456" })).status, + ).toBe(204); + + // Both the old session and the old password are invalidated. + expect((await rexApi.get("/api/me")).status).toBe(401); + await expect(loginUser(t.app, "rex", "password-123")).rejects.toThrow(); + const again = await loginUser(t.app, "rex", "password-456"); + expect(again.user.passwordIsInitial).toBe(true); + }); + + it("删除用户:owned Project(含目录)连带删除,成员关系随级联清除", async () => { + const gone = await provisionUser(t.app, "gone"); + const goneApi = apiClient(t.app, gone.cookie); + expect( + (await goneApi.post("/api/projects", { projectId: "gone-extra", name: "多余项目" })).status, + ).toBe(201); + // gone is also a member of admin's default_project. + expect( + (await admin.post("/api/projects/default_project/members", { userId: "gone" })).status, + ).toBe(201); + + expect((await admin.delete("/api/admin/users/gone")).status).toBe(204); + + // Session and account are invalidated; entry disappears from the list. + expect((await goneApi.get("/api/me")).status).toBe(401); + await expect(loginUser(t.app, "gone", "password-123")).rejects.toThrow(); + const list = (await (await admin.get("/api/admin/users")).json()) as AdminUsersResponse; + expect(list.users.map((u) => u.userId)).toEqual(["admin"]); + + // Both the DB row and the directory for the owned Project are gone. + const rows = t.deps.db.prepare("SELECT project_id FROM projects ORDER BY project_id").all(); + expect(rows.map((r) => r.project_id)).toEqual(["default_project"]); + await expect(fs.access(path.join(t.root, "gone-default_project"))).rejects.toThrow(); + await expect(fs.access(path.join(t.root, "gone-extra"))).rejects.toThrow(); + + // The membership on default_project has been cascade-cleared. + const members = (await ( + await admin.get("/api/projects/default_project/members") + ).json()) as MembersResponse; + expect(members.members.map((m) => m.userId)).toEqual(["admin"]); + }); + + it("内置 admin 不可删除;未知用户 404", async () => { + expect((await admin.delete("/api/admin/users/admin")).status).toBe(409); + expect((await admin.delete("/api/admin/users/ghost")).status).toBe(404); + }); +}); diff --git a/packages/server/test/agent-trace-detail.test.ts b/packages/server/test/agent-trace-detail.test.ts new file mode 100644 index 0000000..f658d10 --- /dev/null +++ b/packages/server/test/agent-trace-detail.test.ts @@ -0,0 +1,104 @@ +/** + * Agent-level Trace detail endpoint integration tests (FD-3): + * The Trace page directory tree comes from an Agent-level scan (including unmanaged + * subagent child Sessions / Sessions created by the CLI), but the Session-level detail + * endpoint looks up the sessions table, so unmanaged ones return 404 directly. The new + * Agent-level detail endpoint locates the Trace file directly by + * (projectId, agentId, sessionId), with access controlled via requireProjectAccess. + */ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { requestBegin, requestEnd, sessionMeta, userText } from "@prismshadow/penguin-core"; +import type { SessionMetaPayload } from "@prismshadow/penguin-core"; +import type { + ProjectCreateResponse, + TraceAnalysisResponse, + TraceEventsResponse, +} from "../src/api/types.js"; +import { apiClient, createTestApp, provisionUser, writeTraceFile } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +/** An unmanaged child Session (only written to Trace, not inserted into the sessions table). */ +const UNMANAGED = "session-2026-07-06-09-00-00-cafe0001"; + +function metaPayload(): SessionMetaPayload { + return { + session_id: UNMANAGED, + model_id: "sub-model", + provider: "custom", + model_context_window: 1000, + system_prompt: "", + tools: [], + thinking_level: "default", + agent_state: "/tmp/a", + workspace: "/tmp/w", + }; +} + +describe("agent-trace-detail", () => { + let t: TestApp; + let owner: ReturnType; + let outsider: ReturnType; + let projectId: string; + const base = () => `/api/projects/${projectId}/agents/default_agent/traces`; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner"); + const b = await provisionUser(t.app, "outsider"); + owner = apiClient(t.app, a.cookie); + outsider = apiClient(t.app, b.cookie); + const created = (await ( + await owner.post("/api/projects", { projectId: "owner-trace", name: "项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + await writeTraceFile(t.root, projectId, "default_agent", "2026-07-06", UNMANAGED, 1, [ + sessionMeta(metaPayload()), + userText("子会话输入"), + requestBegin(), + requestEnd("completed"), + ]); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("未纳管 Session 经 Session 级端点 404,经 Agent 级明细端点可读", async () => { + // Session-level: no such row in the sessions table -> 404. + expect((await owner.get(`/api/sessions/${UNMANAGED}/traces/1`)).status).toBe(404); + // Agent-level detail: locates the Trace file directly -> 200. + const res = await owner.get(`${base()}/${UNMANAGED}/1`); + expect(res.status).toBe(200); + const body = (await res.json()) as TraceEventsResponse; + expect(body.total).toBe(4); + expect((body.events[1]!.payload as { text: string }).text).toBe("子会话输入"); + }); + + it("Agent 级明细分页参数生效", async () => { + const res = await owner.get(`${base()}/${UNMANAGED}/1?offset=1&limit=2`); + expect(res.status).toBe(200); + const body = (await res.json()) as TraceEventsResponse; + expect(body.offset).toBe(1); + expect(body.limit).toBe(2); + expect(body.events).toHaveLength(2); + // Invalid index / limit -> 400. + expect((await owner.get(`${base()}/${UNMANAGED}/0`)).status).toBe(400); + expect((await owner.get(`${base()}/${UNMANAGED}/1?limit=5000`)).status).toBe(400); + }); + + it("Agent 级性能分析端点派生 Request 配对", async () => { + const res = await owner.get(`${base()}/${UNMANAGED}/1/analysis`); + expect(res.status).toBe(200); + const body = (await res.json()) as TraceAnalysisResponse; + expect(body.requests).toHaveLength(1); + expect(body.requests[0]!.status).toBe("completed"); + }); + + it("不存在的 index → 404 trace_not_found", async () => { + expect((await owner.get(`${base()}/${UNMANAGED}/9`)).status).toBe(404); + }); + + it("无访问权的用户 → 404(requireProjectAccess)", async () => { + expect((await outsider.get(`${base()}/${UNMANAGED}/1`)).status).toBe(404); + expect((await outsider.get(`${base()}/${UNMANAGED}/1/analysis`)).status).toBe(404); + }); +}); diff --git a/packages/server/test/agent-transfer.test.ts b/packages/server/test/agent-transfer.test.ts new file mode 100644 index 0000000..83f8f4a --- /dev/null +++ b/packages/server/test/agent-transfer.test.ts @@ -0,0 +1,131 @@ +/** + * Agent State export/import integration tests: export auto-packages (excluding .vault.toml), any member can export, only the owner can + * import, importing the same or an older version requires confirmation (409 -> succeeds + * after confirm), import replaces agent_state while keeping the current vault, invalid + * packages return 400, and snapshots are written to snapshots/v.tar.gz. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import * as tar from "tar"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { agentStateDir, snapshotsDir } from "@prismshadow/penguin-core"; +import type { + AgentImportResponse, + ProjectCreateResponse, + VaultResponse, +} from "../src/api/types.js"; +import { apiClient, createTestApp, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +describe("agent export/import", () => { + let t: TestApp; + let owner: ReturnType; + let member: ReturnType; + let projectId: string; + let base: string; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner_a"); + const b = await provisionUser(t.app, "member_b"); + owner = apiClient(t.app, a.cookie); + member = apiClient(t.app, b.cookie); + const created = (await ( + await owner.post("/api/projects", { projectId: "owner_a-snap", name: "快照项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + base = `/api/projects/${projectId}/agents/default_agent`; + expect( + (await owner.post(`/api/projects/${projectId}/members`, { userId: "member_b" })).status, + ).toBe(201); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("导出→改动→导入回滚:同版本需确认、agent_state 被替换、vault 保留", async () => { + // First set a vault key (must be preserved after import). + expect( + (await owner.put(`${base}/vault`, { entries: [{ key: "TOKEN", value: "secret-1" }] })).status, + ).toBe(200); + + // Members can export; the snapshot is auto-packaged and written to snapshots/v1.tar.gz. + const exportRes = await member.get(`${base}/export`); + expect(exportRes.status).toBe(200); + expect(exportRes.headers.get("content-type")).toBe("application/gzip"); + expect(exportRes.headers.get("content-disposition")).toContain("default_agent-v1.tar.gz"); + const archive = Buffer.from(await exportRes.arrayBuffer()); + expect(archive.byteLength).toBeGreaterThan(0); + const snapFile = path.join(snapshotsDir(t.root, projectId, "default_agent"), "v1.tar.gz"); + await expect(fs.access(snapFile)).resolves.toBeUndefined(); + // The snapshot excludes the vault (rough gzip check: the archive contents don't contain the plaintext value). + expect(archive.includes(Buffer.from("secret-1"))).toBe(false); + + // Leave a change marker in agent_state after exporting. + const marker = path.join(agentStateDir(t.root, projectId, "default_agent"), "marker.txt"); + await fs.writeFile(marker, "dirty", "utf8"); + + // Importing the same version (v1): owner only; without confirm -> 409, with confirm it succeeds and replaces agent_state. + const body = { dataBase64: archive.toString("base64") }; + expect((await member.post(`${base}/import`, body)).status).toBe(403); + expect((await owner.post(`${base}/import`, body)).status).toBe(409); + const imported = await owner.post(`${base}/import`, { ...body, confirm: true }); + expect(imported.status).toBe(200); + expect(((await imported.json()) as AgentImportResponse).version).toBe(1); + await expect(fs.access(marker)).rejects.toThrow(); // The marker has been overwritten. + + // The current vault is preserved (not from the package, and not cleared either). + const vault = (await (await owner.get(`${base}/vault`)).json()) as VaultResponse; + expect(vault.entries.map((e) => e.key)).toEqual(["TOKEN"]); + }); + + it("高版本包直接导入并落 version;非法包 400", async () => { + // Bump the current agent_state's version to 3, then export to get the v3 package. + const configPath = path.join( + agentStateDir(t.root, projectId, "default_agent"), + "system_config.yaml", + ); + const yaml = await fs.readFile(configPath, "utf8"); + await fs.writeFile(configPath, yaml.replace(/^version: .*$/m, "version: 3"), "utf8"); + const archive = Buffer.from(await (await owner.get(`${base}/export`)).arrayBuffer()); + + // Roll version back to 1: the v3 package is newer than current, so it imports directly without confirmation. + await fs.writeFile(configPath, yaml.replace(/^version: .*$/m, "version: 1"), "utf8"); + const res = await owner.post(`${base}/import`, { dataBase64: archive.toString("base64") }); + expect(res.status).toBe(200); + expect(((await res.json()) as AgentImportResponse).version).toBe(3); + + expect( + (await owner.post(`${base}/import`, { dataBase64: Buffer.from("junk").toString("base64") })) + .status, + ).toBe(400); + }); + + it("包内混入 .vault.toml 会被忽略,导入后仍是现行 vault", async () => { + expect( + (await owner.put(`${base}/vault`, { entries: [{ key: "TOKEN", value: "current" }] })).status, + ).toBe(200); + + // Manually craft a v9 package containing a forged agent_state/.vault.toml. + const src = path.join(t.root, "crafted"); + await fs.mkdir(path.join(src, "agent_state"), { recursive: true }); + await fs.writeFile( + path.join(src, "agent_state", "system_config.yaml"), + 'system_prompt: "hi"\nversion: 9\n', + "utf8", + ); + await fs.writeFile(path.join(src, "agent_state", ".vault.toml"), 'EVIL = "1"\n', "utf8"); + const crafted = path.join(t.root, "crafted.tar.gz"); + await tar.create({ gzip: true, cwd: src, file: crafted }, ["agent_state"]); + + // v9 is newer than the current v1, so it imports directly without confirmation. + const res = await owner.post(`${base}/import`, { + dataBase64: (await fs.readFile(crafted)).toString("base64"), + }); + expect(res.status).toBe(200); + expect(((await res.json()) as AgentImportResponse).version).toBe(9); + + const vault = (await (await owner.get(`${base}/vault`)).json()) as VaultResponse; + expect(vault.entries.map((e) => e.key)).toEqual(["TOKEN"]); + }); +}); diff --git a/packages/server/test/auth.test.ts b/packages/server/test/auth.test.ts new file mode 100644 index 0000000..e352e87 --- /dev/null +++ b/packages/server/test/auth.test.ts @@ -0,0 +1,180 @@ +/** + * Auth flow integration tests (via app.request() injection): admin seeding / login / + * logout / password change / session / initial Project. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import type { MeResponse, ProjectsResponse } from "../src/api/types.js"; +import { apiClient, createTestApp, loginAdmin, loginUser, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +describe("auth", () => { + let t: TestApp; + + beforeEach(async () => { + t = await createTestApp(); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("未登录访问受保护 API 返回 401", async () => { + const res = await t.app.request("/api/projects"); + expect(res.status).toBe(401); + const body = (await res.json()) as { error: { code: string } }; + expect(body.error.code).toBe("unauthorized"); + }); + + it("不开放注册:register 接口不存在", async () => { + // Not logged in: no such route under /api/auth, falls into the protected-section 401. + const anon = await t.app.request("/api/auth/register", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ userId: "alice", password: "password-123" }), + }); + expect(anon.status).toBe(401); + // Logged in: falls through to notFound -> 404, proving the route has indeed been removed. + const admin = await loginAdmin(t.app); + const res = await apiClient(t.app, admin.cookie).post("/api/auth/register", { + userId: "alice", + password: "password-123", + }); + expect(res.status).toBe(404); + }); + + it("种子 admin 纳管 default_project;初始密码带标记", async () => { + const admin = await loginAdmin(t.app); + expect(admin.user.isAdmin).toBe(true); + expect(admin.user.passwordIsInitial).toBe(true); + const api = apiClient(t.app, admin.cookie); + const projects = (await (await api.get("/api/projects")).json()) as ProjectsResponse; + expect(projects.projects.map((p) => p.projectId)).toContain("default_project"); + expect(projects.projects[0]!.role).toBe("owner"); + // default_agent has been initialized (directory exists). + await expect( + fs.access(path.join(t.root, "default_project", "agents", "default_agent", "agent_state")), + ).resolves.toBeUndefined(); + // Seeding is idempotent: re-seeding does not create a duplicate account. + await t.deps.authService.seedAdmin(); + expect(t.deps.db.prepare("SELECT COUNT(*) AS n FROM users").get()?.n).toBe(1); + }); + + it("管理员建号:默认 Project 为 <用户名>-default_project,显示名缺省为用户名", async () => { + const bob = await provisionUser(t.app, "bob"); + expect(bob.user.isAdmin).toBe(false); + expect(bob.user.passwordIsInitial).toBe(true); + const api = apiClient(t.app, bob.cookie); + const projects = (await (await api.get("/api/projects")).json()) as ProjectsResponse; + expect(projects.projects).toHaveLength(1); + const p = projects.projects[0]!; + expect(p.projectId).toBe("bob-default_project"); + expect(p.name).toBe("bob"); + expect(p.role).toBe("owner"); + expect(p.ownerUserId).toBe("bob"); + // The initial Project's .project_config.toml carries the display name and preset model config (the default model is written along with it). + const toml = await fs.readFile( + path.join(t.root, "bob-default_project", ".project_config.toml"), + "utf8", + ); + expect(toml).toContain('name = "bob"'); + expect(toml).toContain( + 'default_model = { provider = "deepseek", model_id = "deepseek-v4-pro" }', + ); + }); + + it("登录 / me / 登出闭环;错误密码 401", async () => { + await provisionUser(t.app, "carol"); + const wrong = await t.app.request("/api/auth/login", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ userId: "carol", password: "wrong-password" }), + }); + expect(wrong.status).toBe(401); + + const { cookie } = await loginUser(t.app, "carol", "password-123"); + expect(cookie.startsWith("penguin_session=")).toBe(true); + + const api = apiClient(t.app, cookie); + const me = (await (await api.get("/api/me")).json()) as MeResponse; + expect(me.user.userId).toBe("carol"); + + const logout = await api.post("/api/auth/logout"); + expect(logout.status).toBe(204); + const after = await api.get("/api/me"); + expect(after.status).toBe(401); + }); + + it("本人改密:旧密码校验、新密码生效、初始密码标记清除", async () => { + const { cookie } = await provisionUser(t.app, "dave"); + const api = apiClient(t.app, cookie); + + const wrongOld = await api.put("/api/me/password", { + oldPassword: "not-the-password", + newPassword: "new-password-1", + }); + expect(wrongOld.status).toBe(400); + const tooShort = await api.put("/api/me/password", { + oldPassword: "password-123", + newPassword: "short", + }); + expect(tooShort.status).toBe(400); + + const ok = await api.put("/api/me/password", { + oldPassword: "password-123", + newPassword: "new-password-1", + }); + expect(ok.status).toBe(204); + // After the password change, the current session remains valid and the initial-password flag is cleared. + const me = (await (await api.get("/api/me")).json()) as MeResponse; + expect(me.user.passwordIsInitial).toBe(false); + // The old password is invalidated, and the new password can log in. + const oldLogin = await t.app.request("/api/auth/login", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ userId: "dave", password: "password-123" }), + }); + expect(oldLogin.status).toBe(401); + await loginUser(t.app, "dave", "new-password-1"); + }); + + it("写请求拒绝非 JSON Content-Type(CSRF 防线)", async () => { + const res = await t.app.request("/api/auth/login", { + method: "POST", + headers: { "content-type": "application/x-www-form-urlencoded" }, + body: "userId=a&password=b", + }); + expect(res.status).toBe(415); + }); + + it("ui prefs 读写", async () => { + const { cookie } = await provisionUser(t.app, "erin"); + const api = apiClient(t.app, cookie); + const empty = (await (await api.get("/api/me/prefs")).json()) as { prefs: unknown }; + expect(empty.prefs).toEqual({}); + await api.put("/api/me/prefs", { theme: "dark", lastProjectId: "default_project" }); + const got = (await (await api.get("/api/me/prefs")).json()) as { + prefs: { theme: string }; + }; + expect(got.prefs.theme).toBe("dark"); + }); + + it("PUT prefs 浅合并,不覆盖其他写入方的字段", async () => { + const { cookie } = await provisionUser(t.app, "fred"); + const api = apiClient(t.app, cookie); + // Simulate two independent writers: switching Project writes lastProjectId, and onboarding writes credentialGuideSeen. + await api.put("/api/me/prefs", { lastProjectId: "p-1" }); + await api.put("/api/me/prefs", { credentialGuideSeen: true }); + const one = (await (await api.get("/api/me/prefs")).json()) as { + prefs: { lastProjectId?: string; credentialGuideSeen?: boolean }; + }; + // The second write must not erase the fields from the first (a prior full replace would drop lastProjectId, causing onboarding to reappear repeatedly). + expect(one.prefs).toEqual({ lastProjectId: "p-1", credentialGuideSeen: true }); + // Switch Project again: credentialGuideSeen is still present. + await api.put("/api/me/prefs", { lastProjectId: "p-2" }); + const two = (await (await api.get("/api/me/prefs")).json()) as { + prefs: { lastProjectId?: string; credentialGuideSeen?: boolean }; + }; + expect(two.prefs).toEqual({ lastProjectId: "p-2", credentialGuideSeen: true }); + }); +}); diff --git a/packages/server/test/authz.test.ts b/packages/server/test/authz.test.ts new file mode 100644 index 0000000..385c6f5 --- /dev/null +++ b/packages/server/test/authz.test.ts @@ -0,0 +1,182 @@ +/** + * Authorization rules integration tests: three perspectives — owner / member / no + * access; permission boundaries for model config and member management; access control + * for Session-level routes via index lookup; default_project deletion protection. + */ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import type { + MembersResponse, + ModelsResponse, + ProjectCreateResponse, + ProjectsResponse, + SessionCreateResponse, +} from "../src/api/types.js"; +import { apiClient, createTestApp, loginAdmin, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +describe("authz", () => { + let t: TestApp; + let owner: ReturnType; + let member: ReturnType; + let outsider: ReturnType; + let projectId: string; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner_a"); + const b = await provisionUser(t.app, "member_b"); + const c = await provisionUser(t.app, "outsider_c"); + owner = apiClient(t.app, a.cookie); + member = apiClient(t.app, b.cookie); + outsider = apiClient(t.app, c.cookie); + const created = (await ( + await owner.post("/api/projects", { projectId: "owner_a-shared", name: "共享项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + const add = await owner.post(`/api/projects/${projectId}/members`, { userId: "member_b" }); + expect(add.status).toBe(201); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("非成员访问返回 404(不泄露存在性)", async () => { + expect((await outsider.get(`/api/projects/${projectId}/models`)).status).toBe(404); + expect((await outsider.get(`/api/projects/${projectId}/agents`)).status).toBe(404); + expect((await outsider.get(`/api/projects/${projectId}/members`)).status).toBe(404); + expect((await outsider.get(`/api/projects/${projectId}/usage`)).status).toBe(404); + }); + + it("member 可读模型(掩码)但不可写;owner 可写", async () => { + const put = await owner.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "custom", modelId: "m-1" }, + models: [{ provider: "custom", modelId: "m-1", apiKey: "sk-super-secret-key-123456" }], + }); + expect(put.status).toBe(200); + + const res = await member.get(`/api/projects/${projectId}/models`); + expect(res.status).toBe(200); + const models = (await res.json()) as ModelsResponse; + expect(models.defaultModel).toEqual({ provider: "custom", modelId: "m-1" }); + const cred = models.models[0]!.credential!; + expect(cred.apiKeyMasked).toBe("sk-s…3456"); + expect(JSON.stringify(models)).not.toContain("sk-super-secret-key-123456"); + + const denied = await member.put(`/api/projects/${projectId}/models`, { + models: [{ provider: "custom", modelId: "m-2" }], + }); + expect(denied.status).toBe(403); + }); + + it("成员管理仅 owner:列表包含 owner 与 member;重复与自授权 409;未知用户 404", async () => { + const list = (await ( + await member.get(`/api/projects/${projectId}/members`) + ).json()) as MembersResponse; + expect(list.members.map((m) => `${m.userId}:${m.role}`)).toEqual([ + "owner_a:owner", + "member_b:member", + ]); + + expect( + (await member.post(`/api/projects/${projectId}/members`, { userId: "outsider_c" })).status, + ).toBe(403); + expect( + (await owner.post(`/api/projects/${projectId}/members`, { userId: "member_b" })).status, + ).toBe(409); + expect( + (await owner.post(`/api/projects/${projectId}/members`, { userId: "owner_a" })).status, + ).toBe(409); + expect( + (await owner.post(`/api/projects/${projectId}/members`, { userId: "ghost" })).status, + ).toBe(404); + + // After removing authorization, the member loses access. + expect((await owner.delete(`/api/projects/${projectId}/members/member_b`)).status).toBe(204); + expect((await member.get(`/api/projects/${projectId}/models`)).status).toBe(404); + }); + + it("Session 级路由经索引反查授权:member 可见、外人 404", async () => { + await owner.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "anthropic", modelId: "claude-sonnet-4-6" }, + models: [{ provider: "anthropic", modelId: "claude-sonnet-4-6" }], + }); + const created = await owner.post( + `/api/projects/${projectId}/agents/default_agent/sessions`, + {}, + ); + expect(created.status).toBe(201); + const { session } = (await created.json()) as SessionCreateResponse; + + expect((await owner.get(`/api/sessions/${session.sessionId}`)).status).toBe(200); + expect((await member.get(`/api/sessions/${session.sessionId}`)).status).toBe(200); + expect((await outsider.get(`/api/sessions/${session.sessionId}`)).status).toBe(404); + expect((await owner.get("/api/sessions/session-unknown")).status).toBe(404); + }); + + it("删除 Project 仅 owner;default_project 与最后一个可访问 Project 拒绝删除", async () => { + expect((await member.delete(`/api/projects/${projectId}`)).status).toBe(403); + expect((await outsider.delete(`/api/projects/${projectId}`)).status).toBe(404); + // default_project is managed by admin: non-owners always get 404, and the owner (admin) is refused with 409. + expect((await owner.delete("/api/projects/default_project")).status).toBe(404); + const admin = apiClient(t.app, (await loginAdmin(t.app)).cookie); + expect((await admin.delete("/api/projects/default_project")).status).toBe(409); + + // outsider-c only has the initial Project created at account setup: deleting it would + // leave the account with no usable Project (stuck on the skeleton screen on the Web + // side) -> refused with 409. + const cProjects = (await (await outsider.get("/api/projects")).json()) as ProjectsResponse; + expect(cProjects.projects).toHaveLength(1); + expect( + (await outsider.delete(`/api/projects/${cProjects.projects[0]!.projectId}`)).status, + ).toBe(409); + + expect((await owner.delete(`/api/projects/${projectId}`)).status).toBe(204); + expect((await owner.get(`/api/projects/${projectId}/models`)).status).toBe(404); + }); + + it("PUT models 整表替换:省略 apiKey 保留、clearApiKey 清除、defaultModel 必须在 models 内", async () => { + await owner.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "custom", modelId: "m-1" }, + models: [ + { + provider: "custom", + modelId: "m-1", + contextWindow: 200000, + pricing: { cacheRead: 0.3, cacheWrite: 3.75, output: 15 }, + apiKey: "sk-original-key-000111", + baseUrl: "https://example.com/v1", + }, + { provider: "custom", modelId: "m-2" }, + ], + }); + // Omitting apiKey: preserves the original credential and created_at. + const second = (await ( + await owner.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "custom", modelId: "m-1" }, + models: [{ provider: "custom", modelId: "m-1", contextWindow: 100000 }], + }) + ).json()) as ModelsResponse; + expect(second.models).toHaveLength(1); // m-2 was removed by the full-table replace + expect(second.models[0]!.credential?.apiKeyMasked).toBe("sk-o…0111"); + expect(second.models[0]!.credential?.baseUrl).toBe("https://example.com/v1"); + expect(second.models[0]!.credential?.createdAt).toBeTruthy(); + expect(second.models[0]!.contextWindow).toBe(100000); + expect(second.models[0]!.pricing).toBeUndefined(); // Fields not resubmitted are removed by the replace + + // clearApiKey + baseUrl null clears the value. + const third = (await ( + await owner.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "custom", modelId: "m-1" }, + models: [{ provider: "custom", modelId: "m-1", clearApiKey: true, baseUrl: null }], + }) + ).json()) as ModelsResponse; + expect(third.models[0]!.credential).toBeUndefined(); + + // defaultModel not present in models -> 400. + const bad = await owner.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "custom", modelId: "ghost" }, + models: [{ provider: "custom", modelId: "m-1" }], + }); + expect(bad.status).toBe(400); + }); +}); diff --git a/packages/server/test/benchmarks.test.ts b/packages/server/test/benchmarks.test.ts new file mode 100644 index 0000000..1530f95 --- /dev/null +++ b/packages/server/test/benchmarks.test.ts @@ -0,0 +1,215 @@ +/** + * Benchmark scoreboard read integration tests (read-only display): benchmark_config.toml title/description and runs + * pass-through (falls back to directory name if missing), scoreboard.yaml v2's + * evaluations[] (summary pass-through, per-case runs array; per-case metrics trust the + * file values, falling back to an average over runs when missing), the legacy format + * (per-case single session_id) parsed as a single run and backfilled, bad entries + * discarded, case count, empty when unconfigured, permissions (members can read, + * outsiders get 404). + * + * Tested with a plain Agent (no sample Benchmark pre-installed); default_agent's sample + * Benchmark assertions live in builtin-agents.test.ts. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { benchmarksDir } from "@prismshadow/penguin-core"; +import type { BenchmarksResponse, ProjectCreateResponse } from "../src/api/types.js"; +import { apiClient, createTestApp, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +const AGENT = "bench_agent"; + +describe("benchmarks api", () => { + let t: TestApp; + let owner: ReturnType; + let member: ReturnType; + let outsider: ReturnType; + let projectId: string; + let base: string; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner_a"); + const b = await provisionUser(t.app, "member_b"); + const c = await provisionUser(t.app, "outsider_c"); + owner = apiClient(t.app, a.cookie); + member = apiClient(t.app, b.cookie); + outsider = apiClient(t.app, c.cookie); + const created = (await ( + await owner.post("/api/projects", { projectId: "owner_a-bench", name: "评测项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + // A plain Agent has no sample Benchmark pre-installed (only default_agent provides one). + expect((await owner.post(`/api/projects/${projectId}/agents`, { agentId: AGENT })).status).toBe( + 201, + ); + base = `/api/projects/${projectId}/agents/${AGENT}/benchmarks`; + expect( + (await owner.post(`/api/projects/${projectId}/members`, { userId: "member_b" })).status, + ).toBe(201); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("未配置时返回空列表", async () => { + expect((await (await owner.get(base)).json()) as BenchmarksResponse).toEqual({ + benchmarks: [], + }); + }); + + it("scoreboard v2:summary 与题级 runs 透传,三指标信任文件、缺失由 runs 平均", async () => { + const dir = path.join(benchmarksDir(t.root, projectId, AGENT), "swe-bench-v2"); + await fs.mkdir(path.join(dir, "CASE-001-excel-task", "statement"), { recursive: true }); + await fs.mkdir(path.join(dir, "CASE-002-web-task", "rubric"), { recursive: true }); + await fs.writeFile( + path.join(dir, "benchmark_config.toml"), + `title = "SWE Bench v2"\ndescription = "示例"\nruns = 2\n`, + "utf8", + ); + await fs.writeFile( + path.join(dir, "scoreboard.yaml"), + [ + "evaluations:", + ' - time: "2026-07-16T10:00:00Z"', + " version: 3", + ' provider: "deepseek"', + ' model_id: "deepseek-v4-pro"', + ' summary_title: "系统 Prompt 增加计划步骤"', + ' summary: "两题各跑两次取平均;本轮为系统 Prompt 增加了计划步骤。"', + " score: 8.0", + " cost: 0.05", + " duration_ms: 120000", + " cases:", + // Per-case metrics are all present: trust the file values (no recomputation even if inconsistent with the runs average). + ' - case: "CASE-001-excel-task"', + " score: 4.2", + " cost: 0.02", + " duration_ms: 50000", + " runs:", + " - score: 4.0", + " cost: 0.018", + " duration_ms: 48000", + ' session_id: "session-run-1"', + " - score: 4.5", + " cost: 0.022", + " duration_ms: 52000", + ' session_id: "session-run-2"', + // Per-case metrics are missing: computed as the average over runs. + ' - case: "CASE-002-web-task"', + " runs:", + " - score: 3.0", + " cost: 0.01", + " duration_ms: 30000", + ' session_id: "session-run-3"', + " - score: 4.0", + " cost: 0.03", + " duration_ms: 40000", + ' session_id: "session-run-4"', + ].join("\n"), + "utf8", + ); + + const res = (await (await member.get(base)).json()) as BenchmarksResponse; + const bench = res.benchmarks[0]!; + expect(bench).toMatchObject({ + id: "swe-bench-v2", + title: "SWE Bench v2", + description: "示例", + runs: 2, + caseCount: 2, + }); + // config carries no model reference (the model lives on each evaluation). + expect("modelId" in bench).toBe(false); + expect("provider" in bench).toBe(false); + const evaluation = bench.evaluations[0]!; + // The evaluation entry carries this run's model (as a pair) and a summary title (curve series / title-body are displayed separately). + expect(evaluation.provider).toBe("deepseek"); + expect(evaluation.modelId).toBe("deepseek-v4-pro"); + expect(evaluation.summaryTitle).toBe("系统 Prompt 增加计划步骤"); + expect(evaluation.summary).toBe("两题各跑两次取平均;本轮为系统 Prompt 增加了计划步骤。"); + expect(evaluation.score).toBe(8.0); + // Per-case metrics are all present: trust the file (4.2, not the runs average of 4.25). + const full = evaluation.cases.find((c) => c.case === "CASE-001-excel-task")!; + expect(full.score).toBe(4.2); + expect(full.cost).toBe(0.02); + expect(full.durationMs).toBe(50000); + expect(full.runs).toEqual([ + { score: 4.0, cost: 0.018, durationMs: 48000, sessionId: "session-run-1" }, + { score: 4.5, cost: 0.022, durationMs: 52000, sessionId: "session-run-2" }, + ]); + // Per-case metrics are missing: computed as the average over runs. + const derived = evaluation.cases.find((c) => c.case === "CASE-002-web-task")!; + expect(derived.score).toBe(3.5); + expect(derived.cost).toBeCloseTo(0.02, 10); + expect(derived.durationMs).toBe(35000); + expect(derived.runs).toHaveLength(2); + expect(derived.sessionId).toBeUndefined(); + }); + + it("旧格式(题级单 session_id)按单次运行解析补一条 run;坏条目丢弃;外人 404", async () => { + const dir = path.join(benchmarksDir(t.root, projectId, AGENT), "swe-bench-v1"); + await fs.mkdir(path.join(dir, "CASE-001-excel-task", "statement"), { recursive: true }); + await fs.writeFile(path.join(dir, "benchmark_config.toml"), `title = "SWE Bench v1"\n`, "utf8"); + await fs.writeFile( + path.join(dir, "scoreboard.yaml"), + [ + "evaluations:", + ' - time: "2026-07-16T10:00:00Z"', + " version: 1", + " score: 62.5", + " cost: 1.25", + " duration_ms: 60000", + " cases:", + ' - case: "CASE-001-excel-task"', + " score: 30", + " cost: 0.5", + " duration_ms: 20000", + ' session_id: "session-abc"', + ' - case: ""', // Bad entry: discarded + " score: 1", + " - time: 42", // Bad evaluation: discarded + " score: 1", + ].join("\n"), + "utf8", + ); + // A benchmark with no config file: title falls back to the directory name, config runs field is absent by default. + await fs.mkdir(path.join(benchmarksDir(t.root, projectId, AGENT), "empty-bench"), { + recursive: true, + }); + + const res = (await (await member.get(base)).json()) as BenchmarksResponse; + expect(res.benchmarks.map((b) => b.id)).toEqual(["empty-bench", "swe-bench-v1"]); + const bench = res.benchmarks[1]!; + expect(bench).toMatchObject({ title: "SWE Bench v1", caseCount: 1 }); + expect("runs" in bench).toBe(false); + expect(bench.evaluations).toHaveLength(1); + expect(bench.evaluations[0]).toMatchObject({ + time: "2026-07-16T10:00:00Z", + version: 1, + score: 62.5, + cost: 1.25, + durationMs: 60000, + }); + expect("summary" in bench.evaluations[0]!).toBe(false); + // Legacy per-case format: fields unchanged, plus one backfilled run matching the case-level values (the frontend uniformly expands via runs). + expect(bench.evaluations[0]?.cases).toEqual([ + { + case: "CASE-001-excel-task", + score: 30, + cost: 0.5, + durationMs: 20000, + sessionId: "session-abc", + runs: [{ score: 30, cost: 0.5, durationMs: 20000, sessionId: "session-abc" }], + }, + ]); + expect(res.benchmarks[0]).toMatchObject({ + title: "empty-bench", + caseCount: 0, + evaluations: [], + }); + + expect((await outsider.get(base)).status).toBe(404); + }); +}); diff --git a/packages/server/test/builtin-agents.test.ts b/packages/server/test/builtin-agents.test.ts new file mode 100644 index 0000000..17f9c1f --- /dev/null +++ b/packages/server/test/builtin-agents.test.ts @@ -0,0 +1,122 @@ +/** + * Project built-in Agent provisioning: the only built-in Agent is default_agent + * (pre-installed with every Skill in the library, empty AGENTS.md, cannot be deleted). + * Specialized capabilities are now carried by Skills — agent_creator / agent_optimizer + * are no longer built-in Agents: neither provisioned nor deletion-protected. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { loadLibrarySkills } from "@prismshadow/penguin-skills"; +import type { BenchmarksResponse } from "../src/api/types.js"; +import { apiClient, createTestApp, provisionUser, type TestApp } from "./helpers.js"; + +interface AgentsResponse { + agents: Array<{ agentId: string; name?: string; description?: string }>; +} +interface ProjectsResponse { + projects: Array<{ projectId: string }>; +} +interface ProjectCreateResponse { + project: { projectId: string }; +} + +describe("内置 Agent 供给", () => { + let t: TestApp; + let owner: ReturnType; + + beforeEach(async () => { + t = await createTestApp(); + const reg = await provisionUser(t.app, "owner1"); + owner = apiClient(t.app, reg.cookie); + }); + afterEach(async () => { + await t.cleanup(); + }); + + async function expectBuiltinAgents(projectId: string): Promise { + const list = (await ( + await owner.get(`/api/projects/${projectId}/agents`) + ).json()) as AgentsResponse; + const ids = list.agents.map((a) => a.agentId); + // The only built-in Agent: default_agent is listed first with a display name; specialized Agents are no longer provisioned. + expect(ids[0]).toBe("default_agent"); + expect(ids).not.toContain("agent_creator"); + expect(ids).not.toContain("agent_optimizer"); + expect(list.agents.find((a) => a.agentId === "default_agent")?.name).toBe("General Agent"); + + // Install policy: default_agent is pre-installed with every Skill currently in the library. + const skillsOf = async (agentId: string) => + ( + await fs.readdir(path.join(t.root, projectId, "agents", agentId, "agent_state", "skills")) + ).sort(); + expect(await skillsOf("default_agent")).toEqual(loadLibrarySkills().map((skill) => skill.name)); + + // The default AGENTS.md is empty: it carries no preset guidance (delegation and task + // conventions live in the default template's Suggested workflows section). + const defaultMd = await fs.readFile( + path.join(t.root, projectId, "agents", "default_agent", "agent_state", "AGENTS.md"), + "utf8", + ); + expect(defaultMd).toBe(""); + } + + it("建号时的初始 Project 自带 default_agent", async () => { + const projects = (await (await owner.get("/api/projects")).json()) as ProjectsResponse; + await expectBuiltinAgents(projects.projects[0]!.projectId); + }); + + it("新建 Project 同样自带 default_agent", async () => { + const created = (await ( + await owner.post("/api/projects", { projectId: "owner1-new", name: "新项目" }) + ).json()) as ProjectCreateResponse; + await expectBuiltinAgents(created.project.projectId); + }); + + it("default_agent 附带示例 Benchmark:GET /benchmarks 可读(3 evaluations、runs=2、summary 存在)", async () => { + const projects = (await (await owner.get("/api/projects")).json()) as ProjectsResponse; + const projectId = projects.projects[0]!.projectId; + const res = await owner.get(`/api/projects/${projectId}/agents/default_agent/benchmarks`); + expect(res.status).toBe(200); + const body = (await res.json()) as BenchmarksResponse; + const bench = body.benchmarks.find((b) => b.id === "example-benchmark")!; + expect(bench).toBeDefined(); + expect(bench.title).toBe("Example Benchmark"); + // description explicitly states it's a built-in sample (the whole directory can be deleted or replaced). + expect(bench.description).toContain("example"); + expect(bench.runs).toBe(2); + expect(bench.caseCount).toBe(2); + expect(bench.evaluations).toHaveLength(3); + for (const evaluation of bench.evaluations) { + expect(evaluation.summary).toBeTruthy(); + expect(evaluation.cases).toHaveLength(2); + for (const c of evaluation.cases) { + expect(c.runs).toHaveLength(2); + for (const run of c.runs!) { + expect(run.sessionId).toMatch(/^session-/); + } + } + } + // The sample data tells an optimization story: scores increase across evaluation rounds (the evaluation center shows a rising curve out of the box). + const scores = bench.evaluations.map((e) => e.score); + expect(scores).toEqual([...scores].sort((a, b) => a - b)); + }); + + it("default_agent 不可删除(409)", async () => { + const projects = (await (await owner.get("/api/projects")).json()) as ProjectsResponse; + const projectId = projects.projects[0]!.projectId; + const res = await owner.delete(`/api/projects/${projectId}/agents/default_agent`); + expect(res.status).toBe(409); + }); + + it("agent_creator 等旧专用 id 不再受内置保护:可自建、可删除", async () => { + const projects = (await (await owner.get("/api/projects")).json()) as ProjectsResponse; + const projectId = projects.projects[0]!.projectId; + const created = await owner.post(`/api/projects/${projectId}/agents`, { + agentId: "agent_creator", + }); + expect(created.status).toBe(201); + const res = await owner.delete(`/api/projects/${projectId}/agents/agent_creator`); + expect(res.status).toBe(204); + }); +}); diff --git a/packages/server/test/channel.test.ts b/packages/server/test/channel.test.ts new file mode 100644 index 0000000..0d0fdf5 --- /dev/null +++ b/packages/server/test/channel.test.ts @@ -0,0 +1,140 @@ +/** + * SSE channel unit tests: epoch-string ids (`-`), ring buffer (dual + * count/byte caps), Last-Event-ID replay hit determination (cross-epoch always misses), + * private sends bypass the buffer, idle reclamation (active channels are skipped). + */ +import { describe, expect, it } from "vitest"; +import { Channel, ChannelHub } from "../src/runtime/channel.js"; +import type { ChannelEvent } from "../src/runtime/channel.js"; + +/** A valid epoch prefix guaranteed to differ from ch.epoch (for building a cross-epoch Last-Event-ID). */ +function foreignEpoch(ch: Channel): string { + return ch.epoch === "00000000" ? "11111111" : "00000000"; +} + +describe("channel", () => { + it("publish 分配 `-` 单调 id 并广播", () => { + const ch = new Channel(); + const seen: ChannelEvent[] = []; + ch.subscribe((e) => seen.push(e)); + ch.publish({ a: 1 }); + ch.publish({ b: 2 }, "server_event"); + expect(ch.epoch).toMatch(/^[0-9a-f]{8}$/); + expect(seen.map((e) => e.id)).toEqual([`${ch.epoch}-1`, `${ch.epoch}-2`]); + expect(seen[0]!.event).toBeUndefined(); + expect(seen[1]!.event).toBe("server_event"); + expect(JSON.parse(seen[0]!.data)).toEqual({ a: 1 }); + }); + + it("退订后不再接收", () => { + const ch = new Channel(); + const seen: ChannelEvent[] = []; + const unsub = ch.subscribe((e) => seen.push(e)); + ch.publish("x"); + unsub(); + ch.publish("y"); + expect(seen).toHaveLength(1); + }); + + it("按条数上限淘汰最旧事件", () => { + const ch = new Channel({ maxBufferCount: 3 }); + for (let i = 0; i < 5; i++) ch.publish(`m${i}`); // seq 1..5, buffer retains 3, 4, 5 + const miss = ch.replayAfter(`${ch.epoch}-1`); // event 2 has been evicted + expect(miss.hit).toBe(false); + const hit = ch.replayAfter(`${ch.epoch}-3`); + expect(hit.hit).toBe(true); + expect(hit.events.map((e) => e.id)).toEqual([`${ch.epoch}-4`, `${ch.epoch}-5`]); + }); + + it("按字节上限淘汰最旧事件", () => { + const ch = new Channel({ maxBufferBytes: 30 }); + ch.publish("a".repeat(20)); + ch.publish("b".repeat(20)); // event 1 was evicted due to the byte cap + expect(ch.replayAfter(`${ch.epoch}-0`).hit).toBe(false); // a subscriber that never saw event 1 needs a resync + const hit = ch.replayAfter(`${ch.epoch}-1`); // has seen event 1 (its eviction doesn't matter) -> hit, replays 2 + expect(hit.hit).toBe(true); + expect(hit.events.map((e) => e.id)).toEqual([`${ch.epoch}-2`]); + }); + + it("Last-Event-ID 等于最新 id → 命中且无补发;超过已分配 seq → miss", () => { + const ch = new Channel(); + ch.publish("x"); // seq 1 + const same = ch.replayAfter(`${ch.epoch}-1`); + expect(same.hit).toBe(true); + expect(same.events).toEqual([]); + expect(ch.replayAfter(`${ch.epoch}-99`).hit).toBe(false); + }); + + it("跨纪元(通道回收重建 / 进程重启)一律 miss——即使 seq 落在命中区间", () => { + const ch = new Channel(); + for (let i = 0; i < 5; i++) ch.publish(`m${i}`); // seq 1..5, all within the buffer + // Same seq, different epoch: the old implementation would false-hit on integer ranges; now it must miss -> resync. + expect(ch.replayAfter(`${foreignEpoch(ch)}-2`).hit).toBe(false); + expect(ch.replayAfter(`${ch.epoch}-2`).hit).toBe(true); + }); + + it("非法 Last-Event-ID(无纪元 / 非整数 seq)→ miss", () => { + const ch = new Channel(); + ch.publish("x"); + expect(ch.replayAfter("42").hit).toBe(false); // legacy pure-integer ids are also treated as unknown + expect(ch.replayAfter("").hit).toBe(false); + expect(ch.replayAfter(`${ch.epoch}-abc`).hit).toBe(false); + expect(ch.replayAfter("-1").hit).toBe(false); + }); + + it("全新通道(无事件)带任何 Last-Event-ID → miss;`-0` 从头命中", () => { + const ch = new Channel(); + expect(ch.replayAfter(`${ch.epoch}-5`).hit).toBe(false); + expect(ch.replayAfter(`${ch.epoch}-0`).hit).toBe(true); // 0 = from the start, hits since nothing was evicted (no events to replay) + }); + + it("sendTo 私发:占用 seq、不进缓冲、不广播", () => { + const ch = new Channel(); + const broadcast: ChannelEvent[] = []; + const priv: ChannelEvent[] = []; + ch.subscribe((e) => broadcast.push(e)); + const evt = ch.sendTo((e) => priv.push(e), { type: "hello" }, "server_event"); + expect(evt.id).toBe(`${ch.epoch}-1`); + expect(priv).toHaveLength(1); + expect(broadcast).toHaveLength(0); + const next = ch.publish("x"); + expect(next.id).toBe(`${ch.epoch}-2`); // the seq number was consumed by the private send + expect(ch.replayAfter(`${ch.epoch}-1`).events.map((e) => e.id)).toEqual([`${ch.epoch}-2`]); // private sends bypass the buffer + }); + + it("hub:同 key 复用通道;空闲超时回收、有订阅者不回收", () => { + const hub = new ChannelHub({ idleMs: 1000 }); + const ch = hub.get("s1"); + expect(hub.get("s1")).toBe(ch); + ch.publish("x"); + const now = Date.now(); + hub.sweep(now + 500); + expect(hub.peek("s1")).toBe(ch); + hub.sweep(now + 2000); + expect(hub.peek("s1")).toBeUndefined(); + + const ch2 = hub.get("s2"); + ch2.subscribe(() => {}); + hub.sweep(Date.now() + 10_000); + expect(hub.peek("s2")).toBe(ch2); // not reclaimed while it has a subscriber + hub.dispose(); + }); + + it("hub:isActive 谓词命中的通道不回收(运行中等待审批无 publish 也保留)", () => { + const active = new Set(["busy"]); + const hub = new ChannelHub({ idleMs: 1000, isActive: (key) => active.has(key) }); + const busy = hub.get("busy"); + const idle = hub.get("idle"); + busy.publish("x"); + idle.publish("x"); + const now = Date.now(); + hub.sweep(now + 10_000); + expect(hub.peek("busy")).toBe(busy); // active: reclamation is skipped + expect(hub.peek("idle")).toBeUndefined(); + // Once back to idle, it's reclaimed under the normal rule. + active.clear(); + hub.sweep(now + 20_000); + expect(hub.peek("busy")).toBeUndefined(); + hub.dispose(); + }); +}); diff --git a/packages/server/test/config.test.ts b/packages/server/test/config.test.ts new file mode 100644 index 0000000..c980ad6 --- /dev/null +++ b/packages/server/test/config.test.ts @@ -0,0 +1,29 @@ +/** + * resolveServerConfig PORT parsing tests: both the default (missing) and empty string + * (the common `PORT=` empty value in `.env`) fall back to 7364 — Number("") === 0 used + * to make the empty string pass range validation and bind to a random port; explicit + * "0" is preserved (explicit semantics for a random available port); invalid values + * throw. This matches the CLI's resolvePort semantics (packages/cli serve). + */ +import { describe, expect, it } from "vitest"; +import { resolveServerConfig } from "../src/config.js"; + +const base = { PENGUIN_HOME: "/tmp/penguin-config-test" }; + +describe("resolveServerConfig:PORT 解析", () => { + it("缺省 7364;空串视为未设置(不落到端口 0)", () => { + expect(resolveServerConfig({ ...base }).port).toBe(7364); + expect(resolveServerConfig({ ...base, PORT: "" }).port).toBe(7364); + }); + + it('显式数值生效;显式 "0" 保留(绑随机可用端口)', () => { + expect(resolveServerConfig({ ...base, PORT: "8930" }).port).toBe(8930); + expect(resolveServerConfig({ ...base, PORT: "0" }).port).toBe(0); + }); + + it("非整数或超界抛错", () => { + for (const bad of ["abc", "3.14", "-1", "65536"]) { + expect(() => resolveServerConfig({ ...base, PORT: bad }), bad).toThrow(/非法端口/); + } + }); +}); diff --git a/packages/server/test/errors.test.ts b/packages/server/test/errors.test.ts new file mode 100644 index 0000000..46d904a --- /dev/null +++ b/packages/server/test/errors.test.ts @@ -0,0 +1,896 @@ +/** + * Error-record persistence unit and integration tests: ErrorsRepo's aggregation semantics (including cross-tenant isolation + * where unattributed errors are **visible only to admins**) and its row-cap eviction + * (evicts the oldest by id, without misfiring on id gaps left by deleteByProject); + * ErrorRecorder's expected/unexpected determination (explicit kind takes priority, HTTP + * infers from HttpError), short-window deduplication (storm protection), and the + * "never throws itself" guarantee; StreamErrorWatcher picking up LLM / Environment + * errors from the message stream (attributed to **the Session that actually produced + * the error**: a child Session's failure is attributed to the child Agent / child + * Session); HTTP onError actually persisting records; cascading cleanup on Project + * deletion. + */ +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import type { DatabaseSync } from "node:sqlite"; +import { + abortEvent, + assistantText, + partialToolCallOutput, + requestBegin, + requestEnd, + sessionMeta, + toolCall, + toolCallOutput, + withOrigin, +} from "@prismshadow/penguin-core"; +import type { OmniMessage } from "@prismshadow/penguin-core"; +import type { ProjectCreateResponse, UsageResponse } from "../src/api/types.js"; +import { openDatabase } from "../src/db/database.js"; +import { ErrorsRepo } from "../src/db/repos/errors.js"; +import type { ErrorRecordInsert } from "../src/db/repos/errors.js"; +import { HttpError } from "../src/http/errors.js"; +import { + DEDUP_KEYS_MAX, + DEDUP_WINDOW_MS, + ErrorRecorder, + MESSAGE_MAX, +} from "../src/runtime/error-recorder.js"; +import { StreamErrorWatcher } from "../src/runtime/stream-error-watcher.js"; +import { apiClient, createTestApp, loginAdmin, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +function row(date: string, o: Partial = {}): ErrorRecordInsert { + return { + ts: `${date}T10:00:00.000Z`, + date, + projectId: "p1", + agentId: null, + sessionId: null, + source: "http", + kind: "unexpected", + code: "internal", + status: 500, + message: "boom", + ...o, + }; +} + +describe("errors-repo", () => { + let db: DatabaseSync; + let repo: ErrorsRepo; + + beforeEach(() => { + db = openDatabase(":memory:"); + repo = new ErrorsRepo(db); + }); + afterEach(() => db.close()); + + it("汇总:总数与其中未预期数;expected 照记不丢", () => { + repo.insert(row("2026-07-06")); + repo.insert(row("2026-07-06", { kind: "expected", code: "not_found", status: 404 })); + repo.insert(row("2026-07-06", { kind: "expected", code: "bad_request", status: 400 })); + expect(repo.summary("p1")).toEqual({ total: 3, unexpected: 1 }); + }); + + it("无归属异常(登录失败 / 进程崩溃)只对管理员可见:普通成员只看本 Project", () => { + const global = { projectId: null, source: "process", code: "uncaught_exception" }; + repo.insert(row("2026-07-06", global)); // Unattributed: another tenant's login failure / process crash + repo.insert(row("2026-07-06", global)); + repo.insert(row("2026-07-06", { projectId: "p-other" })); // Another Project: invisible to everyone + repo.insert(row("2026-07-06", { kind: "expected", code: "not_found", status: 404 })); // This Project + + // Regular member (default includeGlobal=false): all three queries see only the row for this Project. + expect(repo.summary("p1")).toEqual({ total: 1, unexpected: 0 }); + expect(repo.topCode("p1")).toMatchObject({ code: "not_found", count: 1 }); + expect(repo.recent("p1").map((r) => r.code)).toEqual(["not_found"]); + + // Admin: this Project + unattributed (still can't see another Project's rows). + const admin = { includeGlobal: true }; + expect(repo.summary("p1", admin)).toEqual({ total: 3, unexpected: 2 }); + expect(repo.topCode("p1", admin)).toMatchObject({ code: "uncaught_exception", count: 2 }); + expect(repo.recent("p1", admin).map((r) => r.code)).toEqual([ + "not_found", + "uncaught_exception", + "uncaught_exception", + ]); + + // A member of another Project likewise only sees their own row: unattributed errors never land in any regular member's view. + expect(repo.summary("p-other")).toEqual({ total: 1, unexpected: 1 }); + expect(repo.recent("p-other").map((r) => r.code)).toEqual(["internal"]); + }); + + it("最常见错误码:按 source+code+kind 分组取次数最高的一条", () => { + for (let i = 0; i < 3; i++) repo.insert(row("2026-07-06", { code: "internal" })); + repo.insert(row("2026-07-06", { source: "session", code: "session_run_failed" })); + repo.insert(row("2026-07-06", { kind: "expected", code: "not_found", status: 404 })); + repo.insert(row("2026-07-06", { kind: "expected", code: "not_found", status: 404 })); + + expect(repo.topCode("p1")).toEqual({ + source: "http", + code: "internal", + kind: "unexpected", + count: 3, + }); + // No errors / no errors in range -> null (the frontend uses this to hide the metric). + expect(repo.topCode("p-empty")).toBeNull(); + expect(repo.topCode("p1", { from: "2026-07-07" })).toBeNull(); + }); + + it("日期区间与 agent 过滤(HTTP / 进程级异常无 agent_id,按 Agent 过滤时天然只剩该 Agent)", () => { + repo.insert(row("2026-07-05")); + repo.insert(row("2026-07-06", { kind: "expected" })); + repo.insert(row("2026-07-06", { agentId: "a1", source: "session" })); + + expect(repo.summary("p1")).toEqual({ total: 3, unexpected: 2 }); + expect(repo.summary("p1", { from: "2026-07-06" })).toEqual({ total: 2, unexpected: 1 }); + expect(repo.summary("p1", { agentId: "a1" })).toEqual({ total: 1, unexpected: 1 }); + expect(repo.topCode("p1", { agentId: "a1" })).toMatchObject({ source: "session", count: 1 }); + expect(repo.recent("p1", { agentId: "a1" })).toHaveLength(1); + }); + + it("最近异常:时间倒序取前 limit 条", () => { + repo.insert(row("2026-07-05", { message: "旧" })); + repo.insert(row("2026-07-06", { message: "新" })); + const recent = repo.recent("p1", {}, 1); + expect(recent).toHaveLength(1); + expect(recent[0]!.message).toBe("新"); + }); + + it("deleteByProject:只删该 Project 的行,无归属异常保留", () => { + repo.insert(row("2026-07-06")); + repo.insert(row("2026-07-06", { projectId: null })); + repo.deleteByProject("p1"); + const rows = db.prepare("SELECT project_id FROM error_records").all(); + expect(rows).toHaveLength(1); + expect(rows[0]!.project_id).toBeNull(); + }); + + // —— Row cap (the second line of defense against error storms; the first is ErrorRecorder's short-window dedup) —— + + const messages = () => + db + .prepare("SELECT message FROM error_records ORDER BY id") + .all() + .map((r) => r.message as string); + + it("容量上限:超出后按 id 淘汰最旧的行(每 pruneEvery 次插入才检查一次)", () => { + const capped = new ErrorsRepo(db, { maxRows: 5, pruneEvery: 2 }); + for (let i = 0; i < 10; i++) capped.insert(row("2026-07-06", { message: `m${i}` })); + // The 5 most recent rows within the cap are kept, older ones are evicted. + expect(messages()).toEqual(["m5", "m6", "m7", "m8", "m9"]); + }); + + it("淘汰按行数算:deleteByProject 留下的 id 空洞不会误删有效数据(近似式会)", () => { + const capped = new ErrorsRepo(db, { maxRows: 3, pruneEvery: 1 }); + capped.insert(row("2026-07-06", { message: "keep-1" })); // id 1 + capped.insert(row("2026-07-06", { message: "keep-2" })); // id 2 + capped.insert(row("2026-07-06", { projectId: "p-gone", message: "gone" })); // id 3 + capped.deleteByProject("p-gone"); // id 3 becomes a gap: MAX(id) is now decoupled from the actual row count + + // 3 rows in the table = exactly at the cap; none should be deleted. An approximation (id <= MAX(id) - 3) would wrongly delete keep-1. + capped.insert(row("2026-07-06", { message: "keep-3" })); // id 4 + expect(messages()).toEqual(["keep-1", "keep-2", "keep-3"]); + + // Once over the cap, the oldest row is evicted as usual (gaps don't affect the "oldest" determination). + capped.insert(row("2026-07-06", { message: "keep-4" })); // id 5 + expect(messages()).toEqual(["keep-2", "keep-3", "keep-4"]); + }); +}); + +describe("error-recorder", () => { + let db: DatabaseSync; + let repo: ErrorsRepo; + const now = () => new Date("2026-07-06T10:00:00"); + + beforeEach(() => { + db = openDatabase(":memory:"); + repo = new ErrorsRepo(db); + }); + afterEach(() => db.close()); + + it("HttpError → expected(保留 code 与状态码)", () => { + new ErrorRecorder(repo, now).record({ + source: "http", + err: new HttpError(404, "session_not_found", "Session 不存在或无权访问。"), + ctx: { projectId: "p1" }, + }); + const r = db.prepare("SELECT * FROM error_records").get()!; + expect(r.kind).toBe("expected"); + expect(r.code).toBe("session_not_found"); + expect(r.status).toBe(404); + expect(r.project_id).toBe("p1"); + expect(r.date).toBe("2026-07-06"); + }); + + it("非 HttpError → unexpected;HTTP 来源收敛 500,非 HTTP 来源 status 为 NULL", () => { + const rec = new ErrorRecorder(repo, now); + rec.record({ source: "http", err: new Error("boom") }); + rec.record({ + source: "session", + err: new Error("drive 挂了"), + ctx: { projectId: "p1", agentId: "a1", sessionId: "s1" }, + code: "session_run_failed", + }); + const rows = db.prepare("SELECT * FROM error_records ORDER BY id").all(); + expect(rows[0]!.kind).toBe("unexpected"); + expect(rows[0]!.code).toBe("internal"); // Matches the same code convention as handleError's external-facing code + expect(rows[0]!.status).toBe(500); + expect(rows[0]!.project_id).toBeNull(); + expect(rows[1]!.code).toBe("session_run_failed"); + expect(rows[1]!.status).toBeNull(); + expect(rows[1]!.agent_id).toBe("a1"); + expect(rows[1]!.session_id).toBe("s1"); + }); + + it("非 Error 抛出物与超长 message:String 化并截断到上限", () => { + const rec = new ErrorRecorder(repo, now); + rec.record({ source: "process", err: "字符串异常", code: "unhandled_rejection" }); + rec.record({ source: "usage", err: new Error("x".repeat(MESSAGE_MAX + 100)) }); + const rows = db.prepare("SELECT message FROM error_records ORDER BY id").all(); + expect(rows[0]!.message).toBe("字符串异常"); + expect((rows[1]!.message as string).length).toBe(MESSAGE_MAX); + }); + + it("记录器自身出错绝不外抛(否则挂在 onError 上会无限递归)", () => { + const broken = { + insert() { + throw new Error("DB 已关闭"); + }, + } as unknown as ErrorsRepo; + expect(() => + new ErrorRecorder(broken).record({ source: "http", err: new Error("x") }), + ).not.toThrow(); + }); + + it("显式 kind 优先于 HttpError 推断(新来源自报「要不要人介入」)", () => { + const rec = new ErrorRecorder(repo, now); + rec.record({ source: "llm", err: "超时", code: "llm_timeout", kind: "expected" }); + rec.record({ source: "llm", err: "鉴权失败", code: "llm_failed", kind: "unexpected" }); + const rows = db.prepare("SELECT kind, source, status FROM error_records ORDER BY id").all(); + expect(rows[0]).toMatchObject({ kind: "expected", source: "llm", status: null }); + expect(rows[1]).toMatchObject({ kind: "unexpected", source: "llm", status: null }); + }); + + // —— Short-window dedup (the first line of defense against error storms) —— + + const count = () => + db.prepare("SELECT COUNT(*) AS n FROM error_records").get()!.n as unknown as number; + /** Dedup table (private): asserts the hard requirement that it stays "bounded". */ + const lastSeen = (rec: ErrorRecorder) => + (rec as unknown as { lastSeen: Map }).lastSeen; + + it("短窗去重:窗口内的同类异常只落一条,窗口外恢复记录", () => { + let t = Date.parse("2026-07-06T10:00:00Z"); + const rec = new ErrorRecorder(repo, () => new Date(t)); + const boom = () => + rec.record({ + source: "http", + err: new HttpError(404, "not_found", "没有"), + ctx: { projectId: "p1" }, + }); + + boom(); + expect(count()).toBe(1); + + t += DEDUP_WINDOW_MS - 1; // Still within the window: a burst of 404s from a scan discards straight away, no persist + boom(); + boom(); + expect(count()).toBe(1); + + t += 1; // Outside the window: the same kind of error is recorded again (a sustained storm leaves exactly one entry per window, never suppressed forever) + boom(); + expect(count()).toBe(2); + }); + + it("去重不跨 source / code / Project(不同类的异常互不压制)", () => { + const rec = new ErrorRecorder(repo, now); // time frozen: everything lands in the same window + const err = new Error("boom"); + rec.record({ source: "http", err, ctx: { projectId: "p1" }, code: "c1" }); + rec.record({ source: "http", err, ctx: { projectId: "p1" }, code: "c1" }); // same kind: discarded + rec.record({ source: "http", err, ctx: { projectId: "p1" }, code: "c2" }); // different code + rec.record({ source: "http", err, ctx: { projectId: "p2" }, code: "c1" }); // different Project + rec.record({ source: "session", err, ctx: { projectId: "p1" }, code: "c1" }); // different source + rec.record({ source: "http", err, code: "c1" }); // unattributed (project_id is NULL): counts as its own kind + expect(count()).toBe(5); + }); + + it("去重表有界:先清过期项,仍超限则整体清空——清空后照常去重与记录", () => { + let t = Date.parse("2026-07-06T10:00:00Z"); + const rec = new ErrorRecorder(repo, () => new Date(t)); + const boom = (code: string) => + rec.record({ source: "http", err: "boom", ctx: { projectId: "p1" }, code }); + + for (let i = 0; i < DEDUP_KEYS_MAX; i++) boom(`c${i}`); // fill it up (one key per code) + expect(lastSeen(rec).size).toBe(DEDUP_KEYS_MAX); + + t += DEDUP_WINDOW_MS; // all old keys expired: the next entry triggers cleanup, leaving only the newly registered one + boom("after-window"); + expect(lastSeen(rec).size).toBe(1); + + for (let i = 0; i < DEDUP_KEYS_MAX; i++) boom(`d${i}`); // all within the same window: nothing to clean → wipe the whole table + expect(lastSeen(rec).size).toBeLessThanOrEqual(DEDUP_KEYS_MAX); + + // Works normally after being wiped: new errors are still recorded, and duplicates within the window are still discarded. + const before = count(); + boom("tail"); + boom("tail"); + expect(count()).toBe(before + 1); + }); +}); + +describe("stream-error-watcher(LLM / Environment 异常)", () => { + let db: DatabaseSync; + let repo: ErrorsRepo; + const now = () => new Date("2026-07-06T10:00:00"); + const CTX = { projectId: "p1", agentId: "a1", sessionId: "s1" }; + + beforeEach(() => { + db = openDatabase(":memory:"); + repo = new ErrorsRepo(db); + }); + afterEach(() => db.close()); + + const watcher = () => new StreamErrorWatcher(new ErrorRecorder(repo, now), CTX); + const rows = () => + db.prepare("SELECT * FROM error_records ORDER BY id").all() as Array>; + + /** Feeds a sequence of messages and finalizes (close: persists any still-pending failure), returning the persisted rows. */ + function feed(msgs: OmniMessage[]): Array> { + const w = watcher(); + for (const m of msgs) w.observe(m); + w.close(); + return rows(); + } + + /** + * A sub-session's session_meta (its first message): origin = the child Session + * id, and agentId is derived from the parent directory name in the `agent_state` + * path (consistent with SessionManager.registerChildSession). + */ + const childMeta = (sessionId: string, agentState: string) => + withOrigin( + sessionMeta({ + session_id: sessionId, + model_id: "m1", + provider: "custom", + model_context_window: 100000, + system_prompt: "", + tools: [], + thinking_level: "default", + agent_state: agentState, + workspace: "/tmp/w", + }), + sessionId, + ); + + // —— LLM —— + + it("LLM failed → unexpected(不可重试,需人介入);message 取随后 abort 事件的真实原因", () => { + const got = feed([ + requestBegin(), + requestEnd("failed"), + abortEvent("llm request error: 401 invalid api key"), + ]); + expect(got).toHaveLength(1); + expect(got[0]).toMatchObject({ + source: "llm", + kind: "unexpected", + code: "llm_failed", + message: "llm request error: 401 invalid api key", + project_id: "p1", + agent_id: "a1", + session_id: "s1", + status: null, + }); + }); + + it("LLM timeout / malformed → expected(引擎自动重连重试);重试耗尽时 message 取 abort 原因", () => { + const got = feed([ + requestBegin(), + requestEnd("timeout"), // First attempt times out → the engine retries (revealed by the next request_begin: no reason text yet) + requestBegin(), + requestEnd("malformed"), + abortEvent("malformed response failed after 2 retries"), + ]); + expect(got).toHaveLength(2); + expect(got[0]).toMatchObject({ source: "llm", kind: "expected", code: "llm_timeout" }); + expect(got[0]!.message).toContain("超时"); // No abort arrived: falls back to the status text + expect(got[1]).toMatchObject({ + source: "llm", + kind: "expected", + code: "llm_malformed", + message: "malformed response failed after 2 retries", + }); + }); + + it("aborted(用户点「停止」)不是异常:不记录;completed 亦不记", () => { + expect( + feed([ + requestBegin(), + requestEnd("completed"), + requestBegin(), + requestEnd("aborted"), + abortEvent("aborted by user"), + ]), + ).toHaveLength(0); + }); + + it("用户在重试退避期间中断:timeout 是真实失败照记,但 message 不采信用户中断文案", () => { + const got = feed([ + requestBegin(), + requestEnd("timeout"), + abortEvent("aborted during reconnect backoff"), + ]); + expect(got).toHaveLength(1); + expect(got[0]).toMatchObject({ code: "llm_timeout", kind: "expected" }); + expect(got[0]!.message).toContain("超时"); + expect(got[0]!.message).not.toContain("aborted"); + }); + + it("失败先挂起等原因;run 结束仍未揭晓 → close 兜底落库(用 status 文案)", () => { + const w = watcher(); + w.observe(requestBegin()); + w.observe(requestEnd("failed")); + expect(rows()).toHaveLength(0); // Pending: waiting for the abort that immediately follows to supply the real reason + w.close(); + const got = rows(); + expect(got).toHaveLength(1); + expect(got[0]).toMatchObject({ code: "llm_failed", kind: "unexpected" }); + expect(got[0]!.message).toContain("LLM 请求失败"); + }); + + it("父/子会话的 LLM 失败按 origin 各自挂起,abort 原因不串味", () => { + const got = feed([ + requestBegin(), // parent session initiates + withOrigin(requestBegin(), "session-child"), + withOrigin(requestEnd("timeout"), "session-child"), + withOrigin(abortEvent("reconnect failed after 2 retries"), "session-child"), + requestEnd("failed"), // the parent session's failure only wraps up now + abortEvent("llm request error: 500 upstream"), + ]); + expect(got).toHaveLength(2); + expect(got[0]).toMatchObject({ + code: "llm_timeout", + message: "reconnect failed after 2 retries", + }); + expect(got[1]).toMatchObject({ + code: "llm_failed", + message: "llm request error: 500 upstream", // not stolen by the sub-session's abort + }); + }); + + // —— Environment (tool execution) —— + + const call = (name: string, id: string) => toolCall({ name, arguments: "{}", toolCallId: id }); + + it("工具 failed / timeout → environment + expected,code 带工具名(能看出是哪个工具挂了)", () => { + const got = feed([ + call("exec_command", "tc-1"), + toolCallOutput({ + output: "ls: /nope: No such file or directory\n[tool error] exit code 2", + toolCallId: "tc-1", + stopReason: "failed", + }), + call("read_file", "tc-2"), + toolCallOutput({ + output: "[tool timeout: exceeded 30000ms]", + toolCallId: "tc-2", + stopReason: "timeout", + }), + ]); + expect(got).toHaveLength(2); + expect(got[0]).toMatchObject({ + source: "environment", + kind: "expected", // error fed back to the model; the Agent adjusts on its own — no human needed + code: "tool_failed:exec_command", + project_id: "p1", + agent_id: "a1", + session_id: "s1", + }); + expect(got[0]!.message).toContain("[tool error] exit code 2"); // the actual error text + expect(got[1]).toMatchObject({ code: "tool_timeout:read_file", kind: "expected" }); + }); + + it("工具 aborted(拒批 / 用户中断)与 completed 不记", () => { + expect( + feed([ + call("exec_command", "tc-1"), + toolCallOutput({ + output: "Tool call denied by user.", + toolCallId: "tc-1", + stopReason: "aborted", + }), + call("read_file", "tc-2"), + toolCallOutput({ output: "ok", toolCallId: "tc-2", stopReason: "completed" }), + ]), + ).toHaveLength(0); + }); + + it("并行工具:tool_call_id → 工具名各自对上(输出乱序到达也不错位)", () => { + const got = feed([ + call("exec_command", "tc-1"), + call("read_file", "tc-2"), + call("write_file", "tc-3"), + toolCallOutput({ output: "boom-2", toolCallId: "tc-2", stopReason: "failed" }), + toolCallOutput({ output: "ok", toolCallId: "tc-3", stopReason: "completed" }), + toolCallOutput({ output: "boom-1", toolCallId: "tc-1", stopReason: "failed" }), + ]); + expect(got.map((r) => r.code)).toEqual(["tool_failed:read_file", "tool_failed:exec_command"]); + expect(got.map((r) => r.message)).toEqual(["boom-2", "boom-1"]); + }); + + it("子会话(origin)的工具失败照记:工具名与父会话的同名 tool_call_id 不串味", () => { + const got = feed([ + call("exec_command", "tc-1"), // parent session + withOrigin(call("write_file", "tc-1"), "session-child"), // sub-session happens to share the same id + withOrigin( + toolCallOutput({ output: "child boom", toolCallId: "tc-1", stopReason: "failed" }), + "session-child", + ), + toolCallOutput({ output: "parent boom", toolCallId: "tc-1", stopReason: "failed" }), + ]); + expect(got).toHaveLength(2); + expect(got[0]).toMatchObject({ + code: "tool_failed:write_file", // the sub-session's tool name, not overwritten by the parent's tc-1 + message: "child boom", + session_id: "s1", // this test didn't feed the sub-session's session_meta → attribution falls back to the parent ctx (see the "attribution" test cases below) + }); + expect(got[1]).toMatchObject({ code: "tool_failed:exec_command", message: "parent boom" }); + }); + + it("超长工具输出:message 取尾部(失败原因在末尾)并截断到上限", () => { + const got = feed([ + call("exec_command", "tc-1"), + toolCallOutput({ + output: `${"x".repeat(2000)}\n[tool error] boom`, + toolCallId: "tc-1", + stopReason: "failed", + }), + ]); + const message = got[0]!.message as string; + expect(message.length).toBe(MESSAGE_MAX); + expect(message.startsWith("…")).toBe(true); + expect(message.endsWith("[tool error] boom")).toBe(true); // truncating from the head would cut off the reason entirely + }); + + it("无关消息为 no-op:正文、以及流式 partial_*(完整 tool_call_output 才是收尾)", () => { + expect( + feed([ + assistantText("正常输出"), + call("exec_command", "tc-1"), + partialToolCallOutput({ eventType: "stop", toolCallId: "tc-1", stopReason: "failed" }), + ]), + ).toHaveLength(0); + }); + + // —— Attribution: an error is recorded against **the session that actually produced it** (a sub-session's failure must not be attributed to the parent Agent) —— + + it("子会话的 LLM 失败归到子 Agent / 子 Session(不是父的)", () => { + const got = feed([ + childMeta("session-child", "/data/agents/agent-child/agent_state"), + withOrigin(requestBegin(), "session-child"), + withOrigin(requestEnd("failed"), "session-child"), + withOrigin(abortEvent("llm request error: 401 invalid api key"), "session-child"), + ]); + expect(got).toHaveLength(1); + expect(got[0]).toMatchObject({ + source: "llm", + code: "llm_failed", + message: "llm request error: 401 invalid api key", + agent_id: "agent-child", // derived from the agent_state path; not the parent's a1 + session_id: "session-child", + project_id: "p1", // projectId always takes the parent's (a sub-session is always in the same Project) + }); + }); + + it("子会话的工具失败同样归到子会话(code 仍带工具名)", () => { + const got = feed([ + childMeta("session-child", "/data/agents/agent-child/agent_state"), + withOrigin(call("exec_command", "tc-1"), "session-child"), + withOrigin( + toolCallOutput({ output: "child boom", toolCallId: "tc-1", stopReason: "failed" }), + "session-child", + ), + ]); + expect(got).toHaveLength(1); + expect(got[0]).toMatchObject({ + source: "environment", + code: "tool_failed:exec_command", + message: "child boom", + agent_id: "agent-child", + session_id: "session-child", + project_id: "p1", + }); + }); + + it("父子交错到达:各自归各自;主会话(无 origin)的失败仍归父", () => { + const got = feed([ + requestBegin(), // parent session initiates + childMeta("session-child", "/data/agents/agent-child/agent_state"), + withOrigin(requestBegin(), "session-child"), + withOrigin(requestEnd("timeout"), "session-child"), + withOrigin(abortEvent("reconnect failed after 2 retries"), "session-child"), + withOrigin(call("write_file", "tc-9"), "session-child"), + withOrigin( + toolCallOutput({ output: "child tool boom", toolCallId: "tc-9", stopReason: "failed" }), + "session-child", + ), + call("exec_command", "tc-9"), // parent session happens to share the same id + toolCallOutput({ output: "parent tool boom", toolCallId: "tc-9", stopReason: "failed" }), + requestEnd("failed"), // the parent session's LLM failure only wraps up now + abortEvent("llm request error: 500 upstream"), + ]); + // The sub-session's LLM / tool failures attribute to it, the parent's to the parent — the four entries never mix (each has a distinct code, so short-window dedup doesn't suppress any of them). + expect(got.map((r) => [r.code, r.agent_id, r.session_id])).toEqual([ + ["llm_timeout", "agent-child", "session-child"], + ["tool_failed:write_file", "agent-child", "session-child"], + ["tool_failed:exec_command", "a1", "s1"], + ["llm_failed", "a1", "s1"], + ]); + }); + + it("session_meta 还没到就先失败(边界):回退父 ctx,不崩", () => { + const got = feed([ + withOrigin(requestEnd("failed"), "session-child"), // the sub-session's meta hasn't arrived yet + withOrigin(abortEvent("llm request error: 500"), "session-child"), + ]); + expect(got).toHaveLength(1); + expect(got[0]).toMatchObject({ code: "llm_failed", agent_id: "a1", session_id: "s1" }); + }); + + it("agent_state 路径异常(空串):不登记,回退父 ctx(不写进一个不存在的 agentId)", () => { + const got = feed([ + childMeta("session-child", ""), // path.basename(path.dirname("")) === "." → caught by the defensive check + withOrigin(requestEnd("failed"), "session-child"), + withOrigin(abortEvent("llm request error: 500"), "session-child"), + ]); + expect(got).toHaveLength(1); + expect(got[0]).toMatchObject({ code: "llm_failed", agent_id: "a1", session_id: "s1" }); + }); +}); + +describe("HTTP onError 落库(集成)", () => { + let t: TestApp; + let api: ReturnType; + let projectId: string; + + beforeEach(async () => { + t = await createTestApp(); + const u = await provisionUser(t.app, "err_user"); + api = apiClient(t.app, u.cookie); + const created = (await ( + await api.post("/api/projects", { projectId: "err_user-proj", name: "异常项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + }); + afterEach(async () => { + await t.cleanup(); + }); + + const errorRows = () => + t.deps.db.prepare("SELECT * FROM error_records ORDER BY id").all() as Array< + Record + >; + + it("业务错误(HttpError 404)→ expected,带 code / 状态码 / projectId", async () => { + const res = await api.get(`/api/projects/${projectId}/agents/agent-nope/sessions`); + expect(res.status).toBe(404); + + const rows = errorRows(); + expect(rows).toHaveLength(1); + expect(rows[0]!.source).toBe("http"); + expect(rows[0]!.kind).toBe("expected"); + expect(rows[0]!.code).toBe("agent_not_found"); + expect(rows[0]!.status).toBe(404); + // Taken from the route params, but only when the requester actually has access — see the "HTTP error attribution" test group below. + expect(rows[0]!.project_id).toBe(projectId); + }); + + it("未预期异常(服务层抛普通 Error)→ unexpected + 500", async () => { + // handleError logs the stack trace: silence it in the test so it doesn't clutter output. + const spy = vi.spyOn(console, "error").mockImplementation(() => {}); + t.deps.usageService.query = () => { + throw new Error("查询炸了"); + }; + const res = await api.get(`/api/projects/${projectId}/usage?groupBy=date`); + expect(res.status).toBe(500); + spy.mockRestore(); + + const rows = errorRows(); + expect(rows).toHaveLength(1); + expect(rows[0]!.kind).toBe("unexpected"); + expect(rows[0]!.code).toBe("internal"); + expect(rows[0]!.status).toBe(500); + expect(rows[0]!.message).toBe("查询炸了"); + expect(rows[0]!.project_id).toBe(projectId); + }); + + it("异常经 usage 端点暴露:汇总 / 最常见错误码 / 最近异常(统计中心的统计信息 + 表格)", async () => { + // An error in another Project owned by the same owner: attributed to that Project, not this one's view. + const other = (await ( + await api.post("/api/projects", { projectId: "err_user-proj_2", name: "另一个项目" }) + ).json()) as ProjectCreateResponse; + await api.get(`/api/projects/${other.project.projectId}/agents/agent-nope/sessions`); // 404 + await api.get(`/api/projects/${projectId}/agents/agent-nope/sessions`); // 404 → expected + + const res = await api.get(`/api/projects/${projectId}/usage?groupBy=date`); + const body = (await res.json()) as UsageResponse; + // The entry from another Project doesn't count in this view (only this Project's errors + admin-visible unattributed errors show here). + expect(body.errors.total).toBe(1); + expect(body.errors.unexpected).toBe(0); + expect(body.errors.topCode).toEqual({ + source: "http", + code: "agent_not_found", + kind: "expected", + count: 1, + }); + expect(body.errors.recent[0]).toMatchObject({ source: "http", code: "agent_not_found" }); + }); + + it("无归属异常(登录失败)只对管理员可见:普通成员的统计中心里看不到别的租户的异常", async () => { + // A login failure has no Project context → produces one unattributed error (project_id is NULL). + const bad = await t.app.request("/api/auth/login", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ userId: "err_user", password: "wrong-password" }), + }); + expect(bad.status).toBe(401); + expect(errorRows().filter((r) => r.project_id === null)).toHaveLength(1); + + // A regular member in their own Project: sees none of it. + const plain = await provisionUser(t.app, "plain_user"); + expect(plain.user.isAdmin).toBe(false); + const plainApi = apiClient(t.app, plain.cookie); + const own = (await ( + await plainApi.post("/api/projects", { projectId: "plain_user-proj", name: "普通成员的项目" }) + ).json()) as ProjectCreateResponse; + const plainBody = (await ( + await plainApi.get(`/api/projects/${own.project.projectId}/usage?groupBy=date`) + ).json()) as UsageResponse; + expect(plainBody.errors).toMatchObject({ total: 0, unexpected: 0, topCode: null, recent: [] }); + + // The admin can see it — the category most in need of visibility isn't rendered invisible by the isolation. + const adminApi = apiClient(t.app, (await loginAdmin(t.app)).cookie); + const adminBody = (await ( + await adminApi.get(`/api/projects/default_project/usage?groupBy=date`) + ).json()) as UsageResponse; + expect(adminBody.errors.total).toBe(1); + expect(adminBody.errors.topCode).toMatchObject({ source: "http", code: "invalid_credentials" }); + expect(adminBody.errors.recent[0]).toMatchObject({ code: "invalid_credentials" }); + }); + + it("删除 Project 级联清理该 Project 的异常记录", async () => { + await api.get(`/api/projects/${projectId}/agents/agent-nope/sessions`); + expect(errorRows().filter((r) => r.project_id === projectId)).toHaveLength(1); + + const del = await api.delete(`/api/projects/${projectId}`); + expect(del.status).toBe(204); + expect(errorRows().filter((r) => r.project_id === projectId)).toHaveLength(0); + }); +}); + +describe("HTTP 异常归属(仅在请求方确有该 Project 访问权时)", () => { + let t: TestApp; + /** The built-in admin: unattributed errors are visible only to them. */ + let adminApi: ReturnType; + let adminProjectId: string; + /** The victim Project's owner: a regular user, so their stats center only shows errors attributed to this Project. */ + let ownerApi: ReturnType; + let projectId: string; + + beforeEach(async () => { + t = await createTestApp(); + const admin = await loginAdmin(t.app); + adminApi = apiClient(t.app, admin.cookie); + adminProjectId = ( + (await ( + await adminApi.post("/api/projects", { projectId: "admin_proj", name: "管理员的项目" }) + ).json()) as ProjectCreateResponse + ).project.projectId; + + const owner = await provisionUser(t.app, "owner_user"); + expect(owner.user.isAdmin).toBe(false); + ownerApi = apiClient(t.app, owner.cookie); + projectId = ( + (await ( + await ownerApi.post("/api/projects", { projectId: "owner_user-victim", name: "受害项目" }) + ).json()) as ProjectCreateResponse + ).project.projectId; + }); + afterEach(async () => { + await t.cleanup(); + }); + + const errorRows = () => + t.deps.db.prepare("SELECT * FROM error_records ORDER BY id").all() as Array< + Record + >; + /** The victim Project's stats center (owner's view: only this Project's errors). */ + const ownerErrors = async () => + ( + (await ( + await ownerApi.get(`/api/projects/${projectId}/usage?groupBy=date`) + ).json()) as UsageResponse + ).errors; + /** The admin's stats center (this Project + unattributed errors). */ + const adminErrors = async () => + ( + (await ( + await adminApi.get(`/api/projects/${adminProjectId}/usage?groupBy=date`) + ).json()) as UsageResponse + ).errors; + + it("未登录 → 401:不归属该 Project(否则任何人都能往别人的 Project 里灌异常统计)", async () => { + const res = await t.app.request(`/api/projects/${projectId}/usage?groupBy=date`); + expect(res.status).toBe(401); + + // The requester isn't even logged in: this error must not be pinned to projectId. + // Note: today this also **incidentally** relies on a Hono quirk — `c.req.param()` + // resolves only against "the route the current handler belongs to"; the 401 is + // thrown in the `/api/*` authMiddleware, whose route has no :projectId, so no + // value is available there. The attribution guard removes the dependency on + // that quirk: when not logged in, `c.var.user` is undefined at runtime → unattributed. + const rows = errorRows(); + expect(rows).toHaveLength(1); + expect(rows[0]).toMatchObject({ source: "http", code: "unauthorized", status: 401 }); + expect(rows[0]!.project_id).toBeNull(); + + // The owner's stats center gains nothing; the trace of the unauthorized probe lands in the admin's view — right where it belongs. + expect(await ownerErrors()).toMatchObject({ total: 0, unexpected: 0, topCode: null }); + const admin = await adminErrors(); + expect(admin.total).toBe(1); + expect(admin.recent[0]).toMatchObject({ source: "http", code: "unauthorized" }); + }); + + it("已登录但无权(非成员)→ 404:同样不归属", async () => { + const outsider = await provisionUser(t.app, "outsider"); + const res = await apiClient(t.app, outsider.cookie).get( + `/api/projects/${projectId}/usage?groupBy=date`, + ); + expect(res.status).toBe(404); + + const rows = errorRows(); + expect(rows).toHaveLength(1); + expect(rows[0]).toMatchObject({ source: "http", code: "project_not_found", status: 404 }); + expect(rows[0]!.project_id).toBeNull(); + + expect(await ownerErrors()).toMatchObject({ total: 0, unexpected: 0, topCode: null }); + expect((await adminErrors()).recent[0]).toMatchObject({ code: "project_not_found" }); + }); + + it("有权成员(owner / 被授权 member)触发的业务错误 → 照常归属该 Project", async () => { + // owner: an invalid groupBy → 400. + expect((await ownerApi.get(`/api/projects/${projectId}/usage?groupBy=bogus`)).status).toBe(400); + + // An authorized member: a 404 in the same Project → attributed the same way (the member branch of canAccess). + const member = await provisionUser(t.app, "member_user"); + const added = await ownerApi.post(`/api/projects/${projectId}/members`, { + userId: "member_user", + }); + expect(added.status).toBe(201); + const missing = await apiClient(t.app, member.cookie).get( + `/api/projects/${projectId}/agents/agent-nope/sessions`, + ); + expect(missing.status).toBe(404); + + expect(errorRows().map((r) => [r.code, r.project_id])).toEqual([ + ["bad_request", projectId], + ["agent_not_found", projectId], + ]); + expect(await ownerErrors()).toMatchObject({ total: 2, unexpected: 0 }); + }); + + it("归属判断自身抛异常:onError 不被搞崩(错误响应照常,异常按无归属落库)", async () => { + t.deps.projectService.canAccess = () => { + throw new Error("授权判定炸了"); + }; + const res = await ownerApi.get(`/api/projects/${projectId}/usage?groupBy=bogus`); + expect(res.status).toBe(400); // still the original business error: not turned into a 500, nor an empty response + expect(await res.json()).toMatchObject({ error: { code: "bad_request" } }); + + const rows = errorRows(); + expect(rows).toHaveLength(1); + expect(rows[0]!.code).toBe("bad_request"); + expect(rows[0]!.project_id).toBeNull(); // a failed determination always falls back to unattributed + }); +}); diff --git a/packages/server/test/helpers.ts b/packages/server/test/helpers.ts new file mode 100644 index 0000000..646f350 --- /dev/null +++ b/packages/server/test/helpers.ts @@ -0,0 +1,153 @@ +/** + * Test helpers: a temp directory root + an in-memory DB (":memory:") + injecting + * requests via app.request() + building Trace files. + * None of these tests listen on a port or make real LLM requests. + */ +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import type { Hono } from "hono"; +import type { OmniMessage } from "@prismshadow/penguin-core"; +import { buildAppDeps, createApp } from "../src/app.js"; +import type { AppDeps, BuildDepsOverrides } from "../src/app.js"; +import type { AppEnv } from "../src/auth/middleware.js"; +import { ADMIN_INITIAL_PASSWORD, ADMIN_USER_ID } from "../src/auth/service.js"; +import type { ServerConfig } from "../src/config.js"; +import type { UserInfo } from "../src/api/types.js"; + +export async function makeTempRoot(): Promise { + return fs.mkdtemp(path.join(os.tmpdir(), "penguin-server-test-")); +} + +const DAY_MS = 24 * 60 * 60 * 1000; + +export function testConfig(root: string): ServerConfig { + return { + root, + host: "127.0.0.1", + port: 0, + dbPath: ":memory:", + // Points to a nonexistent directory: static hosting is disabled in tests. + webDist: path.join(root, "__no_web_dist__"), + authSessionTtlMs: 7 * DAY_MS, + authSessionRenewMs: 6 * DAY_MS, + }; +} + +export interface TestApp { + app: Hono; + deps: AppDeps; + root: string; + cleanup(): Promise; +} + +export interface TestAppOptions extends BuildDepsOverrides { + /** Runs before seeding the admin (for scenarios pre-populating a default_project config as the CLI would). */ + beforeSeed?: (root: string) => Promise; +} + +export async function createTestApp(options: TestAppOptions = {}): Promise { + const { beforeSeed, ...overrides } = options; + const root = await makeTempRoot(); + if (beforeSeed) await beforeSeed(root); + const deps = buildAppDeps(testConfig(root), { log: () => {}, ...overrides }); + // Consistent with the startup entrypoint: seed the built-in admin (owning default_project). + await deps.authService.seedAdmin(); + const app = createApp(deps); + return { + app, + deps, + root, + cleanup: async () => { + deps.channels.dispose(); + deps.db.close(); + await fs.rm(root, { recursive: true, force: true }); + }, + }; +} + +/** Logs in and returns the session cookie (`penguin_session=...`). */ +export async function loginUser( + app: Hono, + userId: string, + password: string, +): Promise<{ cookie: string; user: UserInfo }> { + const res = await app.request("/api/auth/login", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ userId, password }), + }); + if (res.status !== 200) { + throw new Error(`登录失败: ${res.status} ${await res.text()}`); + } + const setCookie = res.headers.get("set-cookie"); + if (!setCookie) throw new Error("登录响应缺少 set-cookie"); + const body = (await res.json()) as { user: UserInfo }; + return { cookie: setCookie.split(";")[0]!, user: body.user }; +} + +/** Logs in as the seeded admin. */ +export function loginAdmin(app: Hono): Promise<{ cookie: string; user: UserInfo }> { + return loginUser(app, ADMIN_USER_ID, ADMIN_INITIAL_PASSWORD); +} + +/** Admin creates the account and logs in as that user (the only way to create test users while registration is closed). */ +export async function provisionUser( + app: Hono, + userId: string, + password = "password-123", +): Promise<{ cookie: string; user: UserInfo }> { + if (userId === ADMIN_USER_ID) return loginAdmin(app); + const admin = await loginAdmin(app); + const res = await apiClient(app, admin.cookie).post("/api/admin/users", { userId, password }); + if (res.status !== 201) { + throw new Error(`建号失败: ${res.status} ${await res.text()}`); + } + return loginUser(app, userId, password); +} + +/** JSON request client that carries the cookie. */ +export function apiClient(app: Hono, cookie: string) { + const call = (method: string) => (apiPath: string, body?: unknown) => + app.request(apiPath, { + method, + headers: { + cookie, + ...(body !== undefined ? { "content-type": "application/json" } : {}), + }, + ...(body !== undefined ? { body: JSON.stringify(body) } : {}), + }); + return { + get: (apiPath: string) => app.request(apiPath, { headers: { cookie } }), + post: call("POST"), + put: call("PUT"), + patch: call("PATCH"), + delete: call("DELETE"), + }; +} + +/** Writes a Trace JSONL file directly (for building historical / discovery scenarios). */ +export async function writeTraceFile( + root: string, + projectId: string, + agentId: string, + dateDir: string, + sessionId: string, + index: number, + messages: OmniMessage[], +): Promise { + const dir = path.join(root, projectId, "agents", agentId, "traces", dateDir); + await fs.mkdir(dir, { recursive: true }); + const file = path.join(dir, `${sessionId}_${String(index).padStart(3, "0")}.jsonl`); + await fs.writeFile(file, messages.map((m) => JSON.stringify(m)).join("\n") + "\n", "utf8"); + return file; +} + +/** Simple wait: until the condition is true or it times out. */ +export async function waitFor(cond: () => boolean, timeoutMs = 2000): Promise { + const start = Date.now(); + while (!cond()) { + if (Date.now() - start > timeoutMs) throw new Error("waitFor 超时"); + await new Promise((r) => setTimeout(r, 5)); + } +} diff --git a/packages/server/test/id-validation.test.ts b/packages/server/test/id-validation.test.ts new file mode 100644 index 0000000..7addb25 --- /dev/null +++ b/packages/server/test/id-validation.test.ts @@ -0,0 +1,103 @@ +/** + * Integration tests for path-parameter id validation (FD-4, path + * traversal prevention): Hono decodes URL-encoded `%2F` into a single path + * parameter — a traversal-style agentId (`..//...`), if passed through + * to path construction unchanged, would let an attacker read/write another + * user's Agent config and Trace across Projects. + * Every route that takes :agentId (config / traces / sessions / trace detail) + * must return 404. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import type { ProjectCreateResponse } from "../src/api/types.js"; +import { apiClient, createTestApp, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +describe("id-validation", () => { + let t: TestApp; + let attacker: ReturnType; + let attackerProject: string; + let victimProject: string; + + /** Traversal-style agentId pointing at the victim Project's default_agent (URL-encoded so it all lands in :agentId). */ + const traversal = () => `..%2F${victimProject}%2Fdefault_agent`; + const agentBase = () => `/api/projects/${attackerProject}/agents`; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "attacker"); + const v = await provisionUser(t.app, "victim"); + attacker = apiClient(t.app, a.cookie); + const victim = apiClient(t.app, v.cookie); + const created = (await ( + await attacker.post("/api/projects", { projectId: "attacker-proj", name: "攻击者项目" }) + ).json()) as ProjectCreateResponse; + attackerProject = created.project.projectId; + const victimCreated = (await ( + await victim.post("/api/projects", { projectId: "victim-proj", name: "受害者项目" }) + ).json()) as ProjectCreateResponse; + victimProject = victimCreated.project.projectId; + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("GET config:穿越型 agentId → 404,不泄露他人 system_config.yaml", async () => { + // The victim Agent's config does exist (readable by the victim via the legitimate path). + const victimConfigPath = path.join( + t.root, + victimProject, + "agents", + "default_agent", + "agent_state", + "system_config.yaml", + ); + await expect(fs.access(victimConfigPath)).resolves.toBeUndefined(); + + const res = await attacker.get(`${agentBase()}/${traversal()}/config`); + expect(res.status).toBe(404); + // Legitimate access to one's own Agent is unaffected. + const own = await attacker.get(`${agentBase()}/default_agent/config`); + expect(own.status).toBe(200); + }); + + it("PUT config:穿越型 agentId → 404,受害文件未被覆写", async () => { + const victimConfigPath = path.join( + t.root, + victimProject, + "agents", + "default_agent", + "agent_state", + "system_config.yaml", + ); + const before = await fs.readFile(victimConfigPath, "utf8"); + const res = await attacker.put(`${agentBase()}/${traversal()}/config`, { + config: { systemPrompt: "pwned" }, + }); + expect(res.status).toBe(404); + expect(await fs.readFile(victimConfigPath, "utf8")).toBe(before); + }); + + it("GET traces / GET|POST sessions:穿越型 agentId → 404", async () => { + expect((await attacker.get(`${agentBase()}/${traversal()}/traces`)).status).toBe(404); + expect((await attacker.get(`${agentBase()}/${traversal()}/sessions`)).status).toBe(404); + expect((await attacker.post(`${agentBase()}/${traversal()}/sessions`, {})).status).toBe(404); + }); + + it("Trace 明细端点:穿越型 sessionId / agentId → 404", async () => { + expect((await attacker.get(`${agentBase()}/default_agent/traces/..%2Fx/1`)).status).toBe(404); + expect( + (await attacker.get(`${agentBase()}/default_agent/traces/..%2Fx/1/analysis`)).status, + ).toBe(404); + expect((await attacker.get(`${agentBase()}/${traversal()}/traces/s/1`)).status).toBe(404); + }); + + it("穿越型 / 含非法字符的 projectId → 404(防御性校验)", async () => { + expect((await attacker.get(`/api/projects/..%2Fetc/agents`)).status).toBe(404); + expect((await attacker.get(`/api/projects/..%2Fetc/models`)).status).toBe(404); + expect((await attacker.get(`/api/projects/..%2Fetc/usage`)).status).toBe(404); + expect((await attacker.get(`/api/projects/..%2Fetc/members`)).status).toBe(404); + expect((await attacker.delete(`/api/projects/..%2F${victimProject}`)).status).toBe(404); + }); +}); diff --git a/packages/server/test/models-vision.test.ts b/packages/server/test/models-vision.test.ts new file mode 100644 index 0000000..7f09e21 --- /dev/null +++ b/packages/server/test/models-vision.test.ts @@ -0,0 +1,124 @@ +/** + * Round-trip of a model's vision flag (whether image input is supported) through + * PUT/GET: explicit false is persisted and read back; omission means supported + * (the response carries no field); a non-boolean value returns 400. + */ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import type { ModelsResponse, ProjectCreateResponse } from "../src/api/types.js"; +import { apiClient, createTestApp, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +describe("models vision 标注", () => { + let t: TestApp; + let owner: ReturnType; + let projectId: string; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner_v"); + owner = apiClient(t.app, a.cookie); + const created = (await ( + await owner.post("/api/projects", { projectId: "owner_v-vision", name: "vision 项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("vision=false 落盘回读;目录外模型省略 = 支持(不携带字段)", async () => { + // Use a custom id outside the catalog to verify pure TOML semantics (a catalog id + // falls back to the catalog's own annotation — see the next test case). + const put = await owner.put(`/api/projects/${projectId}/models`, { + models: [ + { provider: "custom", modelId: "blind-model", vision: false }, + { provider: "custom", modelId: "plain-model" }, + ], + }); + expect(put.status).toBe(200); + const body = (await put.json()) as ModelsResponse; + const blind = body.models.find((m) => m.provider === "custom" && m.modelId === "blind-model")!; + const plain = body.models.find((m) => m.provider === "custom" && m.modelId === "plain-model")!; + expect(blind.vision).toBe(false); + expect("vision" in plain).toBe(false); + + // PUT again without vision: whole-table replace semantics clear the annotation + // (back to the default of supported). + const put2 = await owner.put(`/api/projects/${projectId}/models`, { + models: [{ provider: "custom", modelId: "blind-model" }], + }); + const body2 = (await put2.json()) as ModelsResponse; + expect("vision" in body2.models[0]!).toBe(false); + }); + + it("目录内模型无 TOML 标注时 vision 回落内置目录标注", async () => { + const put = await owner.put(`/api/projects/${projectId}/models`, { + models: [ + { provider: "deepseek", modelId: "deepseek-v4-pro" }, + { provider: "google", modelId: "gemini-3-flash-preview" }, + ], + }); + const body = (await put.json()) as ModelsResponse; + expect( + body.models.find((m) => m.provider === "deepseek" && m.modelId === "deepseek-v4-pro")!.vision, + ).toBe(false); + expect( + body.models.find((m) => m.provider === "google" && m.modelId === "gemini-3-flash-preview")! + .vision, + ).toBe(true); + }); + + it("visionModel 指针:往返、省略保留、目标失效即移除", async () => { + const put = await owner.put(`/api/projects/${projectId}/models`, { + visionModel: { provider: "google", modelId: "gemini-3-flash-preview" }, + models: [ + { provider: "deepseek", modelId: "deepseek-v4-pro", vision: false }, + { provider: "google", modelId: "gemini-3-flash-preview" }, + ], + }); + expect(put.status).toBe(200); + expect(((await put.json()) as ModelsResponse).visionModel).toEqual({ + provider: "google", + modelId: "gemini-3-flash-preview", + }); + + // Omitting visionModel: the original value is preserved. + const put2 = await owner.put(`/api/projects/${projectId}/models`, { + models: [ + { provider: "deepseek", modelId: "deepseek-v4-pro", vision: false }, + { provider: "google", modelId: "gemini-3-flash-preview" }, + ], + }); + expect(((await put2.json()) as ModelsResponse).visionModel).toEqual({ + provider: "google", + modelId: "gemini-3-flash-preview", + }); + + // The former vision model is now annotated as not supporting images: the + // annotation takes priority, so the pointer is removed. + const put3 = await owner.put(`/api/projects/${projectId}/models`, { + models: [{ provider: "google", modelId: "gemini-3-flash-preview", vision: false }], + }); + expect("visionModel" in ((await put3.json()) as ModelsResponse)).toBe(false); + }); + + it("visionModel 不在 models 内或指向不支持图片的模型:400", async () => { + const missing = await owner.put(`/api/projects/${projectId}/models`, { + visionModel: { provider: "custom", modelId: "nope" }, + models: [{ provider: "custom", modelId: "m-1" }], + }); + expect(missing.status).toBe(400); + const blind = await owner.put(`/api/projects/${projectId}/models`, { + visionModel: { provider: "custom", modelId: "m-1" }, + models: [{ provider: "custom", modelId: "m-1", vision: false }], + }); + expect(blind.status).toBe(400); + }); + + it("vision 非布尔值 400", async () => { + const bad = await owner.put(`/api/projects/${projectId}/models`, { + models: [{ provider: "custom", modelId: "m-1", vision: "no" }], + }); + expect(bad.status).toBe(400); + }); +}); diff --git a/packages/server/test/models.test.ts b/packages/server/test/models.test.ts new file mode 100644 index 0000000..d40e7ea Binary files /dev/null and b/packages/server/test/models.test.ts differ diff --git a/packages/server/test/password.test.ts b/packages/server/test/password.test.ts new file mode 100644 index 0000000..961fcf1 --- /dev/null +++ b/packages/server/test/password.test.ts @@ -0,0 +1,36 @@ +/** + * Unit tests for scrypt password hashing: format, verification, salt randomness, + * and fallback behavior for invalid stored strings. + */ +import { describe, expect, it } from "vitest"; +import { hashPassword, verifyPassword } from "../src/auth/password.js"; + +describe("password", () => { + it("散列格式为 scrypt$N$r$p$salt$hash 且可校验", async () => { + const stored = await hashPassword("hello-world-123"); + const parts = stored.split("$"); + expect(parts).toHaveLength(6); + expect(parts[0]).toBe("scrypt"); + expect(Number(parts[1])).toBeGreaterThan(0); + await expect(verifyPassword("hello-world-123", stored)).resolves.toBe(true); + }); + + it("错误密码校验失败", async () => { + const stored = await hashPassword("correct-password"); + await expect(verifyPassword("wrong-password", stored)).resolves.toBe(false); + }); + + it("同一密码两次散列结果不同(随机盐)", async () => { + const a = await hashPassword("same-password"); + const b = await hashPassword("same-password"); + expect(a).not.toBe(b); + await expect(verifyPassword("same-password", a)).resolves.toBe(true); + await expect(verifyPassword("same-password", b)).resolves.toBe(true); + }); + + it("非法存储串返回 false 而非抛异常", async () => { + await expect(verifyPassword("x", "not-a-hash")).resolves.toBe(false); + await expect(verifyPassword("x", "bcrypt$a$b$c$d$e")).resolves.toBe(false); + await expect(verifyPassword("x", "scrypt$abc$8$1$!!$!!")).resolves.toBe(false); + }); +}); diff --git a/packages/server/test/schedule-file.test.ts b/packages/server/test/schedule-file.test.ts new file mode 100644 index 0000000..193f804 --- /dev/null +++ b/packages/server/test/schedule-file.test.ts @@ -0,0 +1,112 @@ +/** + * Pure functions for schedule file parsing and trigger-time computation. + */ +import { describe, expect, it } from "vitest"; +import { + latestSlotAt, + MIN_PERIOD_MS, + nextSlotAfter, + parsePeriod, + parseScheduleFile, + slotInWindow, +} from "../src/runtime/schedule-file.js"; + +const BASE = `prompt = "做日报"\nenabled = true\nstart_at = "2026-07-16T09:00:00Z"\n`; + +function defOf(raw: string) { + const r = parseScheduleFile("daily-report", raw); + if (!r.ok) throw new Error(r.error); + return r.def; +} + +describe("parsePeriod", () => { + it("解析 m/h/d 固定间隔", () => { + expect(parsePeriod("30m")).toBe(30 * 60_000); + expect(parsePeriod("12h")).toBe(12 * 3_600_000); + expect(parsePeriod("7d")).toBe(7 * 86_400_000); + }); + it("非法形态返回 null", () => { + for (const bad of ["", "5", "m30", "1.5h", "-1d", "10s", "1w"]) { + expect(parsePeriod(bad)).toBeNull(); + } + }); +}); + +describe("parseScheduleFile", () => { + it("解析全字段并保留时刻原文", () => { + const def = defOf( + `${BASE}period = "30m"\nend_at = "2026-07-17T09:00:00Z"\nsession_id = "session-x"\n`, + ); + expect(def).toMatchObject({ + name: "daily-report", + prompt: "做日报", + enabled: true, + startAt: "2026-07-16T09:00:00Z", + period: "30m", + periodMs: 30 * 60_000, + endAt: "2026-07-17T09:00:00Z", + sessionId: "session-x", + }); + }); + + it("enabled 缺省为不生效;period 缺省即一次性", () => { + const def = defOf(`prompt = "p"\nstart_at = "2026-07-16T09:00:00Z"\n`); + expect(def.enabled).toBe(false); + expect(def.periodMs).toBeUndefined(); + }); + + it("非法文件逐类拒绝", () => { + const cases: Array<[string, string]> = [ + ["not toml ===", "TOML"], + [`enabled = true\nstart_at = "2026-07-16T09:00:00Z"\n`, "prompt"], + [`prompt = "p"\nenabled = "yes"\nstart_at = "2026-07-16T09:00:00Z"\n`, "enabled"], + [`prompt = "p"\nstart_at = "someday"\n`, "start_at"], + [`${BASE}period = "4m"\n`, "下限"], + [`${BASE}period = "10s"\n`, "period"], + [`${BASE}end_at = "2026-07-16T08:00:00Z"\n`, "end_at"], + [`${BASE}session_id = "s"\nworkspace = "/tmp/w"\n`, "新建 Session 模式"], + [`${BASE}session_id = "s"\nmodel_id = "m1"\n`, "新建 Session 模式"], + [`${BASE}model_id = ""\n`, "model_id"], + ]; + for (const [raw, hint] of cases) { + const r = parseScheduleFile("x", raw); + expect(r.ok, raw).toBe(false); + if (!r.ok) expect(r.error).toContain(hint); + } + expect(MIN_PERIOD_MS).toBe(5 * 60_000); + }); +}); + +describe("latestSlotAt / slotInWindow", () => { + const start = Date.parse("2026-07-16T09:00:00Z"); + it("一次性任务的唯一时刻即 start_at", () => { + const def = defOf(BASE); + expect(latestSlotAt(def, start - 1)).toBeNull(); + expect(latestSlotAt(def, start)).toBe(start); + expect(latestSlotAt(def, start + 999_999)).toBe(start); + }); + it("周期任务自 start_at 按 period 步进", () => { + const def = defOf(`${BASE}period = "30m"\n`); + expect(latestSlotAt(def, start - 1)).toBeNull(); + expect(latestSlotAt(def, start)).toBe(start); + expect(latestSlotAt(def, start + 29 * 60_000)).toBe(start); + expect(latestSlotAt(def, start + 61 * 60_000)).toBe(start + 60 * 60_000); + }); + it("nextSlotAfter:严格晚于 now 的下一时刻;一次性到点后无值;越过 end_at 无值", () => { + const oneShot = defOf(BASE); + expect(nextSlotAfter(oneShot, start - 1)).toBe(start); + expect(nextSlotAfter(oneShot, start)).toBeNull(); + const periodic = defOf(`${BASE}period = "30m"\nend_at = "2026-07-16T10:00:00Z"\n`); + expect(nextSlotAfter(periodic, start - 1)).toBe(start); + expect(nextSlotAfter(periodic, start)).toBe(start + 30 * 60_000); + expect(nextSlotAfter(periodic, start + 45 * 60_000)).toBe(start + 60 * 60_000); + expect(nextSlotAfter(periodic, start + 60 * 60_000)).toBeNull(); // next slot passes end_at + }); + + it("end_at 窗口判定", () => { + const def = defOf(`${BASE}period = "30m"\nend_at = "2026-07-16T10:00:00Z"\n`); + expect(slotInWindow(def, start)).toBe(true); + expect(slotInWindow(def, Date.parse("2026-07-16T10:00:00Z"))).toBe(true); + expect(slotInWindow(def, Date.parse("2026-07-16T10:30:00Z"))).toBe(false); + }); +}); diff --git a/packages/server/test/scheduler.test.ts b/packages/server/test/scheduler.test.ts new file mode 100644 index 0000000..6c1a0ea --- /dev/null +++ b/packages/server/test/scheduler.test.ts @@ -0,0 +1,332 @@ +/** + * Scheduler runtime semantics: missed slots are not + * backfilled; a one-shot task registered after its time is marked missed + * immediately; periodic stepping never fires twice for the same slot; a busy + * target queues and sends once idle; a deleted bound Session marks the task + * invalid, recoverable by editing the file; deleting the file clears the run + * state; new-Session mode; event notifications. + * All tests use doubles and a controlled clock — no real LLM / core Session. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { saveProjectConfig, scheduleDir } from "@prismshadow/penguin-core"; +import { openDatabase } from "../src/db/database.js"; +import { ProjectsRepo } from "../src/db/repos/projects.js"; +import { SchedulesRepo } from "../src/db/repos/schedules.js"; +import { SessionsRepo } from "../src/db/repos/sessions.js"; +import { UsersRepo } from "../src/db/repos/users.js"; +import type { ErrorRecordArgs } from "../src/runtime/error-recorder.js"; +import { Scheduler } from "../src/runtime/scheduler.js"; +import type { ScheduleServerEvent } from "../src/api/types.js"; +import { makeTempRoot } from "./helpers.js"; + +const P = "p1"; +const A = "agent_x"; +const T0 = Date.parse("2026-07-16T09:00:00Z"); +const MIN = 60_000; + +describe("scheduler", () => { + let root: string; + let db: ReturnType; + let repo: SchedulesRepo; + let sessions: SessionsRepo; + let nowMs: number; + let busy: Set; + let started: Array<{ sessionId: string; text: string }>; + let created: Array<{ + projectId: string; + agentId: string; + workspace?: string; + provider?: string; + modelId?: string; + }>; + let events: Array<{ userId: string; event: ScheduleServerEvent }>; + let errors: ErrorRecordArgs[]; + let scheduler: Scheduler; + + beforeEach(async () => { + root = await makeTempRoot(); + // A schedule's model reference is resolved against the Project config during + // reconciliation, so provide a minimal config here: + // m-bench is globally unique, so provider can be omitted and still resolve uniquely. + await saveProjectConfig(root, P, { + default_model: { provider: "custom", model_id: "m-bench" }, + models: [{ provider: "custom", model_id: "m-bench" }], + }); + db = openDatabase(":memory:"); + const users = new UsersRepo(db); + users.insert({ + userId: "owner_a", + passwordHash: "x", + isAdmin: false, + passwordIsInitial: false, + createdAt: "2026-07-16T00:00:00Z", + }); + const projects = new ProjectsRepo(db); + projects.insert({ projectId: P, ownerUserId: "owner_a", createdAt: "2026-07-16T00:00:00Z" }); + repo = new SchedulesRepo(db); + sessions = new SessionsRepo(db); + nowMs = T0; + busy = new Set(); + started = []; + created = []; + events = []; + errors = []; + let seq = 0; + scheduler = new Scheduler({ + root, + repo, + projects, + sessions, + runner: { + statusOf: (id) => (busy.has(id) ? "running" : "idle"), + startTask: async (sessionId, input) => { + started.push({ sessionId, text: JSON.stringify(input[0]?.payload ?? "") }); + return { sessionId }; + }, + }, + sessionCreator: { + createSession: async (args) => { + created.push(args); + const sessionId = `session-new-${++seq}`; + insertSession(sessionId); + return { sessionId }; + }, + }, + errors: { record: (args) => void errors.push(args) }, + notify: (userId, event) => void events.push({ userId, event }), + now: () => nowMs, + }); + await fs.mkdir(scheduleDir(root, P, A), { recursive: true }); + }); + afterEach(() => { + scheduler.stop(); + db.close(); + }); + + function insertSession(sessionId: string): void { + sessions.insert({ + sessionId, + projectId: P, + agentId: A, + modelId: "m1", + provider: "custom", + workspace: "/tmp/w", + approvalMode: "allow-all", + title: null, + createdAt: new Date(nowMs).toISOString(), + }); + } + + async function writeFile(name: string, lines: string[]): Promise { + await fs.writeFile( + path.join(scheduleDir(root, P, A), `${name}.toml`), + lines.join("\n"), + "utf8", + ); + } + + function iso(ms: number): string { + return new Date(ms).toISOString(); + } + + it("周期任务:登记消化过去时刻(错过不补),到点触发一次且不重复", async () => { + insertSession("session-1"); + await writeFile("report", [ + `prompt = "报告"`, + `enabled = true`, + `start_at = "${iso(T0 - 60 * MIN)}"`, + `period = "30m"`, + `session_id = "session-1"`, + ]); + await scheduler.tickOnce(); // Registration: all slots before 09:00 are consumed without firing. + expect(started).toHaveLength(0); + + nowMs = T0 + 30 * MIN; // The next slot that should fire (start+90m) + await scheduler.tickOnce(); + expect(started).toHaveLength(1); + expect(started[0]?.sessionId).toBe("session-1"); + // Trigger input = source block + the prompt body (tells the model this is a scheduled task). + expect(started[0]?.text).toContain(""); + expect(started[0]?.text).toContain("schedule: report"); + expect(started[0]?.text).toContain("报告"); + expect(events.map((e) => e.event.type)).toContain("schedule_fired"); + expect(events[0]?.userId).toBe("owner_a"); + + await scheduler.tickOnce(); // The same slot doesn't fire twice. + expect(started).toHaveLength(1); + + const state = repo.find(P, A, "report"); + expect(state?.lastFiredAt).toBe(iso(T0 + 30 * MIN)); + }); + + it("一次性任务:未来到点触发一次;登记时已过期则标记错过且永不触发", async () => { + insertSession("session-1"); + await writeFile("future", [ + `prompt = "f"`, + `enabled = true`, + `start_at = "${iso(T0 + 10 * MIN)}"`, + `session_id = "session-1"`, + ]); + await writeFile("stale", [ + `prompt = "s"`, + `enabled = true`, + `start_at = "${iso(T0 - 10 * MIN)}"`, + `session_id = "session-1"`, + ]); + await scheduler.tickOnce(); + expect(started).toHaveLength(0); + expect(repo.find(P, A, "stale")?.missed).toBe(true); + + nowMs = T0 + 11 * MIN; + await scheduler.tickOnce(); + expect(started).toHaveLength(1); + expect(repo.find(P, A, "future")?.firedOnce).toBe(true); + + nowMs = T0 + 60 * MIN; + await scheduler.tickOnce(); + expect(started).toHaveLength(1); + }); + + it("忙时排队:目标运行中先排队并通知,空闲后补发;重复到点不叠加", async () => { + insertSession("session-1"); + busy.add("session-1"); + await writeFile("q", [ + `prompt = "排队"`, + `enabled = true`, + `start_at = "${iso(T0)}"`, + `period = "5m"`, + `session_id = "session-1"`, + ]); + // Let registration happen before start_at, so the registration baseline + // doesn't consume the first slot. + nowMs = T0 - MIN; + await scheduler.tickOnce(); + nowMs = T0; + await scheduler.tickOnce(); + expect(started).toHaveLength(0); + expect(events.map((e) => e.event.type)).toEqual(["schedule_queued"]); + + nowMs = T0 + 5 * MIN; // Another slot: still busy, so it's just consumed, not re-queued. + await scheduler.tickOnce(); + expect(events.map((e) => e.event.type)).toEqual(["schedule_queued"]); + + busy.delete("session-1"); + nowMs = T0 + 6 * MIN; + await scheduler.tickOnce(); + expect(started).toHaveLength(1); + expect(events.map((e) => e.event.type)).toEqual(["schedule_queued", "schedule_fired"]); + }); + + it("绑定 Session 不存在:记异常并标记失效;文件修改后恢复", async () => { + await writeFile("ghost", [ + `prompt = "g"`, + `enabled = true`, + `start_at = "${iso(T0 - MIN)}"`, + `period = "5m"`, + `session_id = "session-gone"`, + ]); + nowMs = T0 - 10 * MIN; + await scheduler.tickOnce(); // Registration (the first future slot hasn't arrived yet). + nowMs = T0 + 4 * MIN; + await scheduler.tickOnce(); // Slot arrives: Session doesn't exist → invalidated. + expect(errors.some((e) => e.code === "schedule_session_missing")).toBe(true); + expect(repo.find(P, A, "ghost")?.invalidReason).toBe("session_missing"); + + nowMs = T0 + 9 * MIN; + await scheduler.tickOnce(); // No further attempts while invalid. + expect(errors.filter((e) => e.code === "schedule_session_missing")).toHaveLength(1); + + // Editing the file (rebinding to an existing Session): clears the invalid state and resumes firing. + insertSession("session-2"); + await writeFile("ghost", [ + `prompt = "g2"`, + `enabled = true`, + `start_at = "${iso(T0 - MIN)}"`, + `period = "5m"`, + `session_id = "session-2"`, + ]); + nowMs = T0 + 14 * MIN; + await scheduler.tickOnce(); + expect(repo.find(P, A, "ghost")?.invalidReason).toBeNull(); + expect(started.map((s) => s.sessionId)).toEqual(["session-2"]); + }); + + it("enabled=false 不触发;文件删除即清理运行状态", async () => { + insertSession("session-1"); + await writeFile("off", [ + `prompt = "o"`, + `enabled = false`, + `start_at = "${iso(T0 - MIN)}"`, + `period = "5m"`, + `session_id = "session-1"`, + ]); + nowMs = T0 + 30 * MIN; + await scheduler.tickOnce(); + expect(started).toHaveLength(0); + expect(repo.find(P, A, "off")).not.toBeNull(); + + await fs.unlink(path.join(scheduleDir(root, P, A), "off.toml")); + await scheduler.tickOnce(); + expect(repo.find(P, A, "off")).toBeNull(); + }); + + it("新建 Session 模式:每次触发开新会话并发送(透传 workspace)", async () => { + await writeFile("fresh", [ + `prompt = "新会话"`, + `enabled = true`, + `start_at = "${iso(T0 + MIN)}"`, + `period = "5m"`, + `workspace = "/tmp/ws"`, + `model_id = "m-bench"`, + ]); + await scheduler.tickOnce(); + nowMs = T0 + MIN; + await scheduler.tickOnce(); + nowMs = T0 + 6 * MIN; + await scheduler.tickOnce(); + expect(created).toHaveLength(2); + // The file only supplies model_id (no provider): passed through as-is, resolved + // into a paired reference downstream via a unique match. + expect(created[0]).toMatchObject({ + projectId: P, + agentId: A, + workspace: "/tmp/ws", + modelId: "m-bench", + }); + expect("provider" in created[0]!).toBe(false); + expect(started.map((s) => s.sessionId)).toEqual(["session-new-1", "session-new-2"]); + }); + + it("新建 Session 模式:文件给出 provider 时成对透传", async () => { + await writeFile("paired", [ + `prompt = "成对"`, + `enabled = true`, + `start_at = "${iso(T0 + MIN)}"`, + `provider = "custom"`, + `model_id = "m-bench"`, + ]); + await scheduler.tickOnce(); + nowMs = T0 + MIN; + await scheduler.tickOnce(); + expect(created).toHaveLength(1); + expect(created[0]).toMatchObject({ provider: "custom", modelId: "m-bench" }); + }); + + it("非法文件跳过并记异常,不影响其余任务", async () => { + insertSession("session-1"); + await writeFile("bad", [`prompt = "x"`, `enabled = true`, `start_at = "nonsense"`]); + await writeFile("good", [ + `prompt = "g"`, + `enabled = true`, + `start_at = "${iso(T0 + MIN)}"`, + `session_id = "session-1"`, + ]); + await scheduler.tickOnce(); + nowMs = T0 + 2 * MIN; + await scheduler.tickOnce(); + expect(errors.some((e) => e.code === "schedule_invalid_file")).toBe(true); + expect(started).toHaveLength(1); + }); +}); diff --git a/packages/server/test/schedules.test.ts b/packages/server/test/schedules.test.ts new file mode 100644 index 0000000..0e651eb --- /dev/null +++ b/packages/server/test/schedules.test.ts @@ -0,0 +1,178 @@ +/** + * Integration tests for the schedule routes: CRUD and permissions (any member can read, only the owner can + * modify, outsiders get 404), 400 validation, 409 on name collision, hand-edited + * invalid files landing in invalidFiles, expired one-shot tasks marked missed + * during reconciliation, and the on-disk file shape. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { scheduleDir } from "@prismshadow/penguin-core"; +import type { ProjectCreateResponse, ScheduleItem, SchedulesResponse } from "../src/api/types.js"; +import { apiClient, createTestApp, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +const FUTURE = "2099-01-01T09:00:00Z"; + +describe("schedules api", () => { + let t: TestApp; + let owner: ReturnType; + let member: ReturnType; + let outsider: ReturnType; + let projectId: string; + let base: string; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner_a"); + const b = await provisionUser(t.app, "member_b"); + const c = await provisionUser(t.app, "outsider_c"); + owner = apiClient(t.app, a.cookie); + member = apiClient(t.app, b.cookie); + outsider = apiClient(t.app, c.cookie); + const created = (await ( + await owner.post("/api/projects", { projectId: "owner_a-sched", name: "定时任务项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + base = `/api/projects/${projectId}/agents/default_agent/schedules`; + expect( + (await owner.post(`/api/projects/${projectId}/members`, { userId: "member_b" })).status, + ).toBe(201); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("POST 创建(仅 owner)并落盘 TOML;GET 列表与单条;DELETE 清理", async () => { + const body = { + name: "daily-report", + prompt: "写日报", + enabled: true, + startAt: FUTURE, + period: "30m", + }; + expect((await member.post(base, body)).status).toBe(403); + const createdRes = await owner.post(base, body); + expect(createdRes.status).toBe(201); + const item = (await createdRes.json()) as ScheduleItem; + expect(item).toMatchObject({ + name: "daily-report", + prompt: "写日报", + enabled: true, + period: "30m", + status: "active", + queued: false, + creatorUserId: "owner_a", + nextFireAt: "2099-01-01T09:00:00.000Z", + }); + + // The filename is the identifier; the TOML lands under agent_state/schedule/. + const file = path.join(scheduleDir(t.root, projectId, "default_agent"), "daily-report.toml"); + const raw = await fs.readFile(file, "utf8"); + expect(raw).toContain('prompt = "写日报"'); + expect(raw).toContain('period = "30m"'); + + // Any member can read; outsiders get 404. + const list = (await (await member.get(base)).json()) as SchedulesResponse; + expect(list.schedules.map((s) => s.name)).toEqual(["daily-report"]); + expect(list.invalidFiles).toEqual([]); + expect((await outsider.get(base)).status).toBe(404); + expect((await member.get(`${base}/daily-report`)).status).toBe(200); + + // Name collision returns 409. + expect((await owner.post(base, body)).status).toBe(409); + + // Delete (owner only): removes both the file and the state. + expect((await member.delete(`${base}/daily-report`)).status).toBe(403); + expect((await owner.delete(`${base}/daily-report`)).status).toBe(204); + await expect(fs.access(file)).rejects.toThrow(); + expect((await owner.delete(`${base}/daily-report`)).status).toBe(404); + }); + + it("新建 Session 模式可指定 workspace 与成对模型引用并回显", async () => { + // The model reference is given as a pair (provider + modelId), and it must + // resolve within the Project config. + const res = await owner.post(base, { + name: "fresh", + prompt: "p", + enabled: true, + startAt: FUTURE, + workspace: "/tmp/ws", + provider: "deepseek", + modelId: "deepseek-v4-pro", + }); + expect(res.status).toBe(201); + expect((await res.json()) as ScheduleItem).toMatchObject({ + workspace: "/tmp/ws", + provider: "deepseek", + modelId: "deepseek-v4-pro", + }); + }); + + it("模型引用不可解析(配置里没有)即 400,不落盘", async () => { + const res = await owner.post(base, { + name: "bad-model", + prompt: "p", + enabled: true, + startAt: FUTURE, + provider: "custom", + modelId: "not-configured", + }); + expect(res.status).toBe(400); + expect((await owner.get(`${base}/bad-model`)).status).toBe(404); + }); + + it("PUT 整文件替换(仅 owner);不存在 404", async () => { + const body = { name: "job", prompt: "p1", enabled: false, startAt: FUTURE }; + expect((await owner.post(base, body)).status).toBe(201); + expect( + (await member.put(`${base}/job`, { prompt: "p2", enabled: true, startAt: FUTURE })).status, + ).toBe(403); + const updated = await owner.put(`${base}/job`, { + prompt: "p2", + enabled: true, + startAt: FUTURE, + sessionId: "session-2099", + }); + expect(updated.status).toBe(200); + expect((await updated.json()) as ScheduleItem).toMatchObject({ + prompt: "p2", + enabled: true, + sessionId: "session-2099", + status: "active", + }); + expect( + (await owner.put(`${base}/nope`, { prompt: "p", enabled: false, startAt: FUTURE })).status, + ).toBe(404); + }); + + it("校验 400:period 低于下限、时刻非法、目标二选一、任务名非法", async () => { + const ok = { prompt: "p", enabled: false, startAt: FUTURE }; + const cases = [ + { name: "j1", ...ok, period: "4m" }, + { name: "j2", ...ok, startAt: "someday" }, + { name: "j3", ...ok, sessionId: "s", workspace: "/w" }, + { name: "bad.name", ...ok }, + ]; + for (const body of cases) { + expect((await owner.post(base, body)).status, JSON.stringify(body)).toBe(400); + } + }); + + it("手编文件:非法进 invalidFiles;过期一次性任务对账标记 missed", async () => { + const dir = scheduleDir(t.root, projectId, "default_agent"); + await fs.mkdir(dir, { recursive: true }); + await fs.writeFile(path.join(dir, "broken.toml"), "not = toml =", "utf8"); + await fs.writeFile( + path.join(dir, "stale.toml"), + `prompt = "s"\nenabled = true\nstart_at = "2020-01-01T00:00:00Z"\n`, + "utf8", + ); + const list = (await (await owner.get(base)).json()) as SchedulesResponse; + expect(list.invalidFiles.map((f) => f.name)).toEqual(["broken"]); + const stale = list.schedules.find((s) => s.name === "stale"); + expect(stale?.status).toBe("missed"); + // For a hand-edited file, the creator falls back to the Project owner. + expect(stale?.creatorUserId).toBe("owner_a"); + }); +}); diff --git a/packages/server/test/semantic-ids.test.ts b/packages/server/test/semantic-ids.test.ts new file mode 100644 index 0000000..58396c7 --- /dev/null +++ b/packages/server/test/semantic-ids.test.ts @@ -0,0 +1,169 @@ +/** + * Integration tests for semantic ids: Project / Agent id + * is chosen by the creator — must start with a lowercase letter and contain + * only lowercase letters, digits, and underscores; checked for collisions + * against both the DB and the directory (including built-in reserved ids), + * returning 409 when taken. + * The hyphen is a reserved separator: a non-admin's Project id is forced to + * "-"; the admin's Project id contains no hyphen. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import type { AgentCreateResponse, ProjectCreateResponse } from "../src/api/types.js"; +import { apiClient, createTestApp, loginAdmin, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +const BAD_IDS = ["Foo", "1abc", "a", "-abc", "a-b", "a b", "a.b", "项目", "a".repeat(65)]; + +describe("语义 id", () => { + let t: TestApp; + let admin: ReturnType; + let api: ReturnType; + + beforeEach(async () => { + t = await createTestApp(); + admin = apiClient(t.app, (await loginAdmin(t.app)).cookie); + api = apiClient(t.app, (await provisionUser(t.app, "ida")).cookie); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("建 Project(admin 无前缀、不含连字符):非法 id 400;合法 id 落目录,显示名缺省为 id", async () => { + for (const bad of BAD_IDS) { + const res = await admin.post("/api/projects", { projectId: bad, name: "x" }); + expect(res.status, `projectId=${bad}`).toBe(400); + } + + const created = await admin.post("/api/projects", { projectId: "my_proj_2" }); + expect(created.status).toBe(201); + const { project } = (await created.json()) as ProjectCreateResponse; + expect(project.projectId).toBe("my_proj_2"); + expect(project.name).toBe("my_proj_2"); // No display name given: defaults to the id + await expect(fs.access(path.join(t.root, "my_proj_2"))).resolves.toBeUndefined(); + }); + + it("建 Project:DB 占用、纯目录占用与保留 id 都是 409", async () => { + expect((await admin.post("/api/projects", { projectId: "taken", name: "a" })).status).toBe(201); + expect((await admin.post("/api/projects", { projectId: "taken", name: "b" })).status).toBe(409); + // A directory that exists but isn't tracked (e.g. created by the CLI) is also considered taken. + await fs.mkdir(path.join(t.root, "dir_only"), { recursive: true }); + expect((await admin.post("/api/projects", { projectId: "dir_only", name: "c" })).status).toBe( + 409, + ); + // default_project is already tracked by admin. + expect( + (await admin.post("/api/projects", { projectId: "default_project", name: "d" })).status, + ).toBe(409); + }); + + it("非管理员建 Project:id 强制为 <用户名>-<后缀>,后缀仅小写字母数字下划线", async () => { + // No prefix / prefix only / suffix with a hyphen or an invalid character: 400. + for (const bad of ["blog", "ida", "ida-", "idablog", "proj_ida", "ida-sub-x", "ida-Bad"]) { + const res = await api.post("/api/projects", { projectId: bad, name: "x" }); + expect(res.status, `projectId=${bad}`).toBe(400); + const body = (await res.json()) as { error: { code: string } }; + expect(body.error.code, `projectId=${bad}`).toBe("project_id_prefix_required"); + } + // With the prefix: created normally. + const created = await api.post("/api/projects", { projectId: "ida-blog" }); + expect(created.status).toBe(201); + await expect(fs.access(path.join(t.root, "ida-blog"))).resolves.toBeUndefined(); + }); + + it("创建 Project 中途失败:DB 行与目录回滚,同 id 重试可成功", async () => { + // Inject a config write failure (handleError logs the stack trace: silence it so it doesn't clutter output). + const spy = vi.spyOn(console, "error").mockImplementation(() => {}); + const original = t.deps.projectConfigService.writeInitialConfig.bind( + t.deps.projectConfigService, + ); + t.deps.projectConfigService.writeInitialConfig = async () => { + throw new Error("写配置炸了"); + }; + expect((await admin.post("/api/projects", { projectId: "flaky", name: "x" })).status).toBe(500); + spy.mockRestore(); + // No leftovers: neither the directory nor the DB row exist, so the id isn't held by an orphaned directory. + await expect(fs.access(path.join(t.root, "flaky"))).rejects.toThrow(); + expect( + t.deps.db.prepare("SELECT 1 AS x FROM projects WHERE project_id = ?").get("flaky"), + ).toBeUndefined(); + t.deps.projectConfigService.writeInitialConfig = original; + expect((await admin.post("/api/projects", { projectId: "flaky", name: "x" })).status).toBe(201); + }); + + it("创建 Agent 中途失败:目录回滚,同 id 重试可成功", async () => { + const spy = vi.spyOn(console, "error").mockImplementation(() => {}); + const original = t.deps.agentConfigService.updateConfig.bind(t.deps.agentConfigService); + t.deps.agentConfigService.updateConfig = async () => { + throw new Error("写配置炸了"); + }; + expect( + (await api.post("/api/projects/ida-default_project/agents", { agentId: "flaky" })).status, + ).toBe(500); + spy.mockRestore(); + await expect( + fs.access(path.join(t.root, "ida-default_project", "agents", "flaky")), + ).rejects.toThrow(); + t.deps.agentConfigService.updateConfig = original; + expect( + (await api.post("/api/projects/ida-default_project/agents", { agentId: "flaky" })).status, + ).toBe(201); + }); + + it("建 Agent:非法 id 400;合法 id 初始化;重复与内置 id 409;跨 Project 不冲突", async () => { + for (const bad of BAD_IDS) { + const res = await api.post("/api/projects/ida-default_project/agents", { + agentId: bad, + name: "x", + }); + expect(res.status, `agentId=${bad}`).toBe(400); + } + + const created = await api.post("/api/projects/ida-default_project/agents", { + agentId: "crawler", + name: "爬虫", + }); + expect(created.status).toBe(201); + const { agent } = (await created.json()) as AgentCreateResponse; + expect(agent.agentId).toBe("crawler"); + expect(agent.name).toBe("爬虫"); + await expect( + fs.access( + path.join( + t.root, + "ida-default_project", + "agents", + "crawler", + "agent_state", + "system_config.yaml", + ), + ), + ).resolves.toBeUndefined(); + + // Both a duplicate within the same Project and a built-in reserved id are blocked by the collision check. + expect( + ( + await api.post("/api/projects/ida-default_project/agents", { + agentId: "crawler", + name: "y", + }) + ).status, + ).toBe(409); + expect( + ( + await api.post("/api/projects/ida-default_project/agents", { + agentId: "default_agent", + name: "y", + }) + ).status, + ).toBe(409); + + // Agent id uniqueness is scoped to the Project: another Project can reuse the + // same id; the name defaults to the id. + expect((await api.post("/api/projects", { projectId: "ida-other" })).status).toBe(201); + const noName = await api.post("/api/projects/ida-other/agents", { agentId: "crawler" }); + expect(noName.status).toBe(201); + expect(((await noName.json()) as AgentCreateResponse).agent.name).toBe("crawler"); + }); +}); diff --git a/packages/server/test/session-delete-scratchpad.test.ts b/packages/server/test/session-delete-scratchpad.test.ts new file mode 100644 index 0000000..95da9aa --- /dev/null +++ b/packages/server/test/session-delete-scratchpad.test.ts @@ -0,0 +1,99 @@ +/** + * Integration tests for Session deletion cleanup: DELETE /api/sessions/:id + * removes the Session's scratchpad directory (model-generated temp files and + * input images saved to disk for models without image support) in addition + * to its Trace and index row. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { scratchpadDir } from "@prismshadow/penguin-core"; +import type { ProjectCreateResponse, SessionCreateResponse } from "../src/api/types.js"; +import { apiClient, createTestApp, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +async function exists(p: string): Promise { + try { + await fs.access(p); + return true; + } catch { + return false; + } +} + +describe("会话删除清理 scratchpad", () => { + let t: TestApp; + let owner: ReturnType; + let projectId: string; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner_s"); + owner = apiClient(t.app, a.cookie); + const created = (await ( + await owner.post("/api/projects", { + projectId: "owner_s-scratchpad", + name: "scratchpad 项目", + }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + await owner.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "anthropic", modelId: "claude-sonnet-4-6" }, + models: [{ provider: "anthropic", modelId: "claude-sonnet-4-6" }], + }); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("DELETE 会话后其 scratchpad 目录一并删除", async () => { + const created = await owner.post( + `/api/projects/${projectId}/agents/default_agent/sessions`, + {}, + ); + expect(created.status).toBe(201); + const { session } = (await created.json()) as SessionCreateResponse; + + // Simulate temp files written during the session (input images / model-generated files). + const dir = path.join(scratchpadDir(t.root, projectId, "default_agent"), session.sessionId); + await fs.mkdir(dir, { recursive: true }); + await fs.writeFile(path.join(dir, "upload-1.png"), "fake"); + expect(await exists(dir)).toBe(true); + + const del = await owner.delete(`/api/sessions/${session.sessionId}`); + expect(del.status).toBe(204); + expect(await exists(dir)).toBe(false); + // Deleting a session with no scratchpad also succeeds (rm force is idempotent). + const created2 = await owner.post( + `/api/projects/${projectId}/agents/default_agent/sessions`, + {}, + ); + const { session: s2 } = (await created2.json()) as SessionCreateResponse; + expect((await owner.delete(`/api/sessions/${s2.sessionId}`)).status).toBe(204); + }); + + it("GET /sessions/:id/scratchpad/:file 按会话读取文件;缺失或非法文件名 404", async () => { + const created = await owner.post( + `/api/projects/${projectId}/agents/default_agent/sessions`, + {}, + ); + const { session } = (await created.json()) as SessionCreateResponse; + const dir = path.join(scratchpadDir(t.root, projectId, "default_agent"), session.sessionId); + await fs.mkdir(dir, { recursive: true }); + const png = Buffer.from("89504e470d0a1a0a", "hex"); // just needs the PNG magic bytes + await fs.writeFile(path.join(dir, "upload-1.png"), png); + + const res = await owner.get(`/api/sessions/${session.sessionId}/scratchpad/upload-1.png`); + expect(res.status).toBe(200); + expect(res.headers.get("content-type")).toBe("image/png"); + expect(Buffer.from(await res.arrayBuffer())).toEqual(png); + + // Missing files and filenames with path separators/traversal both 404 (no existence leak). + expect((await owner.get(`/api/sessions/${session.sessionId}/scratchpad/nope.png`)).status).toBe( + 404, + ); + expect( + (await owner.get(`/api/sessions/${session.sessionId}/scratchpad/..%2Fsecret.png`)).status, + ).toBe(404); + }); +}); diff --git a/packages/server/test/session-index.test.ts b/packages/server/test/session-index.test.ts new file mode 100644 index 0000000..e8273b4 --- /dev/null +++ b/packages/server/test/session-index.test.ts @@ -0,0 +1,286 @@ +/** + * Integration tests for the Session index: creation (default model / workspace + * guard), listing (DB union Trace directory discovery), PATCH approval mode, + * and createdAt parsing. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { sessionMeta, userText } from "@prismshadow/penguin-core"; +import type { SessionMetaPayload } from "@prismshadow/penguin-core"; +import type { + ProjectCreateResponse, + SessionCreateResponse, + SessionResponse, + SessionsResponse, +} from "../src/api/types.js"; +import { sessionIdCreatedAt } from "../src/services/session-service.js"; +import { apiClient, createTestApp, provisionUser, writeTraceFile } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +describe("session-index", () => { + let t: TestApp; + let api: ReturnType; + let projectId: string; + const base = () => `/api/projects/${projectId}/agents/default_agent/sessions`; + + beforeEach(async () => { + t = await createTestApp(); + const { cookie } = await provisionUser(t.app, "alice"); + api = apiClient(t.app, cookie); + const created = (await ( + await api.post("/api/projects", { projectId: "alice-index", name: "测试项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + }); + afterEach(async () => { + await t.cleanup(); + }); + + async function configureModels(): Promise { + const res = await api.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "anthropic", modelId: "claude-sonnet-4-6" }, + models: [{ provider: "anthropic", modelId: "claude-sonnet-4-6", contextWindow: 128000 }], + }); + expect(res.status).toBe(200); + } + + it("未配置默认模型时创建 Session → 400 no_default_model", async () => { + // A newly created Project comes with a default model preset: first replace + // the whole table to clear it (omitting defaultModel + the original default + // absent from models = removes default_model), then verify the + // no-default-model error path. + const cleared = await api.put(`/api/projects/${projectId}/models`, { + models: [{ provider: "custom", modelId: "m-no-default" }], + }); + expect(cleared.status).toBe(200); + const res = await api.post(base(), {}); + expect(res.status).toBe(400); + const body = (await res.json()) as { error: { code: string } }; + expect(body.error.code).toBe("no_default_model"); + }); + + it("模型没有可用 credential 时创建 Session → 400 model_credential_missing", async () => { + // A model using the OpenAI protocol: the SDK requires a credential as soon as + // the client is constructed. Clear the environment variable key so none is + // available — the error must carry an **error code** (the frontend renders + // localized text from the code, not by parsing the Chinese message). + const prev = process.env.OPENAI_API_KEY; + delete process.env.OPENAI_API_KEY; + try { + await api.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "custom", modelId: "no-key-model" }, + models: [{ provider: "custom", modelId: "no-key-model", clientType: "openai" }], + }); + const res = await api.post(base(), {}); + expect(res.status).toBe(400); + const body = (await res.json()) as { error: { code: string; message: string } }; + expect(body.error.code).toBe("model_credential_missing"); + // The raw SDK message (littered with the env var name) must not leak. + expect(body.error.message).not.toMatch(/OPENAI_API_KEY/); + expect(body.error.message).toContain("no-key-model"); + } finally { + if (prev !== undefined) process.env.OPENAI_API_KEY = prev; + } + }); + + it("创建 Session:缺省自动临时 Workspace、默认 allow-all、进列表", async () => { + await configureModels(); + const res = await api.post(base(), {}); + expect(res.status).toBe(201); + const { session } = (await res.json()) as SessionCreateResponse; + expect(session.sessionId).toMatch(/^session-\d{4}-/); + expect(session.modelId).toBe("claude-sonnet-4-6"); + expect(session.approvalMode).toBe("allow-all"); + expect(session.status).toBe("idle"); + expect(session.hasTrace).toBe(false); + // The temporary Workspace lives inside this Agent's workspaces directory. + expect(session.workspace).toContain( + path.join(projectId, "agents", "default_agent", "workspaces"), + ); + + const list = (await (await api.get(base())).json()) as SessionsResponse; + expect(list.sessions.map((s) => s.sessionId)).toContain(session.sessionId); + }); + + it("显式 Workspace 只要求已存在,可在 Project 目录之外", async () => { + await configureModels(); + const inside = path.join(t.root, projectId, "my-workdir"); + await fs.mkdir(inside, { recursive: true }); + const ok = await api.post(base(), { workspace: inside }); + expect(ok.status).toBe(201); + const { session } = (await ok.json()) as SessionCreateResponse; + expect(session.workspace).toBe(await fs.realpath(inside)); + + // An existing directory outside the Project directory is likewise allowed (reachability is left to file permissions). + const outside = path.join(t.root, "not-a-project"); + await fs.mkdir(outside, { recursive: true }); + const okOutside = await api.post(base(), { workspace: outside }); + expect(okOutside.status).toBe(201); + + // A nonexistent directory is still 400 (not auto-created). + expect( + (await api.post(base(), { workspace: path.join(t.root, projectId, "ghost") })).status, + ).toBe(400); + }); + + it("列表并集:Trace 目录发现未纳管 Session 并补插索引行", async () => { + await configureModels(); + const discovered = "session-2026-07-01-08-30-00-deadbeef"; + const meta: SessionMetaPayload = { + session_id: discovered, + model_id: "cli-model", + provider: "custom", + model_context_window: 1000, + system_prompt: "", + tools: [], + thinking_level: "default", + agent_state: "/tmp/a", + workspace: "/tmp/cli-workspace", + }; + await writeTraceFile(t.root, projectId, "default_agent", "2026-07-01", discovered, 1, [ + sessionMeta(meta), + userText("cli 会话"), + ]); + + const list = (await (await api.get(base())).json()) as SessionsResponse; + const found = list.sessions.find((s) => s.sessionId === discovered); + expect(found).toBeDefined(); + expect(found!.modelId).toBe("cli-model"); + expect(found!.workspace).toBe("/tmp/cli-workspace"); + expect(found!.approvalMode).toBe("allow-all"); + expect(found!.hasTrace).toBe(true); + expect(found!.createdAt).toBe(sessionIdCreatedAt(discovered)); + + // Already indexed: visible via the single-lookup endpoint. + const single = await api.get(`/api/sessions/${discovered}`); + expect(single.status).toBe(200); + }); + + it("DELETE Session:清索引行与全部 Trace 分片,列表不复活;重删 404", async () => { + await configureModels(); + const { session } = (await (await api.post(base(), {})).json()) as SessionCreateResponse; + const sessionId = session.sessionId; + // Create a Trace spanning multiple dated shards: deletion must clear all of + // them, or the listing's directory discovery would resurrect the session. + const meta: SessionMetaPayload = { + session_id: sessionId, + model_id: "anthropic/claude-sonnet-4-6", + provider: "custom", + model_context_window: 1000, + system_prompt: "", + tools: [], + thinking_level: "default", + agent_state: "/tmp/a", + workspace: session.workspace, + }; + const f1 = await writeTraceFile( + t.root, + projectId, + "default_agent", + "2026-07-01", + sessionId, + 1, + [sessionMeta(meta), userText("第一轮")], + ); + const f2 = await writeTraceFile( + t.root, + projectId, + "default_agent", + "2026-07-02", + sessionId, + 2, + [sessionMeta(meta), userText("第二轮")], + ); + + const del = await api.delete(`/api/sessions/${sessionId}`); + expect(del.status).toBe(204); + + await expect(fs.stat(f1)).rejects.toThrow(); + await expect(fs.stat(f2)).rejects.toThrow(); + + const list = (await (await api.get(base())).json()) as SessionsResponse; + expect(list.sessions.map((s) => s.sessionId)).not.toContain(sessionId); + expect((await api.delete(`/api/sessions/${sessionId}`)).status).toBe(404); + expect((await api.get(`/api/sessions/${sessionId}`)).status).toBe(404); + }); + + it("DELETE Session:Workspace 目录不随删除清除(用户自带目录必须保留)", async () => { + await configureModels(); + const inside = path.join(t.root, projectId, "keep-me"); + await fs.mkdir(inside, { recursive: true }); + const { session } = (await ( + await api.post(base(), { workspace: inside }) + ).json()) as SessionCreateResponse; + + expect((await api.delete(`/api/sessions/${session.sessionId}`)).status).toBe(204); + expect((await fs.stat(inside)).isDirectory()).toBe(true); + }); + + it("列表按 createdAt 降序", async () => { + await configureModels(); + const older = "session-2020-01-01-00-00-00-00000001"; + await writeTraceFile(t.root, projectId, "default_agent", "2020-01-01", older, 1, [ + sessionMeta({ + session_id: older, + model_id: "m", + provider: "custom", + model_context_window: 1, + system_prompt: "", + tools: [], + thinking_level: "default", + agent_state: "/a", + workspace: "/w", + }), + ]); + const created = (await (await api.post(base(), {})).json()) as SessionCreateResponse; + const list = (await (await api.get(base())).json()) as SessionsResponse; + expect(list.sessions[0]!.sessionId).toBe(created.session.sessionId); + expect(list.sessions[list.sessions.length - 1]!.sessionId).toBe(older); + }); + + it("PATCH 审批模式即存并回读", async () => { + await configureModels(); + const { session } = (await (await api.post(base(), {})).json()) as SessionCreateResponse; + // Change from the default allow-all to a different mode, to confirm it's actually persisted. + const patched = await api.patch(`/api/sessions/${session.sessionId}`, { + approvalMode: "always-ask", + }); + expect(patched.status).toBe(200); + const got = (await ( + await api.get(`/api/sessions/${session.sessionId}`) + ).json()) as SessionResponse; + expect(got.session.approvalMode).toBe("always-ask"); + // An invalid mode returns 400. + expect( + (await api.patch(`/api/sessions/${session.sessionId}`, { approvalMode: "sometimes" })).status, + ).toBe(400); + }); + + it("insertOrIgnore 幂等:并发首次发现同一 Session 不因 UNIQUE 约束抛错", async () => { + const row = { + sessionId: "session-2026-07-02-00-00-00-11223344", + projectId, + agentId: "default_agent", + modelId: "cli-model", + provider: "custom", + workspace: "/tmp/w", + approvalMode: "always-ask" as const, + title: null, + createdAt: new Date().toISOString(), + }; + t.deps.sessionsRepo.insertOrIgnore(row); + // A second insert with different fields for the same id: silently ignored, no throw, first-inserted value is kept. + expect(() => + t.deps.sessionsRepo.insertOrIgnore({ ...row, modelId: "other-model" }), + ).not.toThrow(); + expect(t.deps.sessionsRepo.findById(row.sessionId)!.modelId).toBe("cli-model"); + }); + + it("sessionIdCreatedAt:非法格式返回 null", () => { + expect(sessionIdCreatedAt("session-2026-07-01-08-30-00-deadbeef")).toBe( + new Date(2026, 6, 1, 8, 30, 0).toISOString(), + ); + expect(sessionIdCreatedAt("not-a-session")).toBeNull(); + }); +}); diff --git a/packages/server/test/session-loader.test.ts b/packages/server/test/session-loader.test.ts new file mode 100644 index 0000000..010ba8a --- /dev/null +++ b/packages/server/test/session-loader.test.ts @@ -0,0 +1,102 @@ +/** + * Integration tests for createCoreSessionLoader (#3/#13): failures recovering a + * historical Session (Workspace deleted / Model removed from config / Trace + * missing session_meta) all collapse into HttpError(409, session_unrecoverable), + * preserving the original core message instead of bubbling up as a 500. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { createAgent, saveProjectConfig, sessionMeta, userText } from "@prismshadow/penguin-core"; +import type { SessionMetaPayload } from "@prismshadow/penguin-core"; +import { createCoreSessionLoader } from "../src/runtime/session-manager.js"; +import type { SessionRow } from "../src/db/repos/sessions.js"; +import { HttpError } from "../src/http/errors.js"; +import { makeTempRoot, writeTraceFile } from "./helpers.js"; + +const PROJECT = "project-loader"; +const AGENT = "default_agent"; +const SID = "session-2026-07-06-11-00-00-abcd0001"; + +function meta(overrides: Partial = {}): SessionMetaPayload { + return { + session_id: SID, + model_id: "custom/m1", + provider: "custom", + model_context_window: 1000, + system_prompt: "sp", + tools: [], + thinking_level: "default", + agent_state: "/tmp/a", + workspace: path.join("/tmp", "does-not-exist-xyz"), + ...overrides, + }; +} + +function row(workspace: string): SessionRow { + return { + sessionId: SID, + projectId: PROJECT, + agentId: AGENT, + modelId: "custom/m1", + provider: "custom", + workspace, + approvalMode: "always-ask", + title: null, + createdAt: new Date().toISOString(), + }; +} + +describe("session-loader", () => { + let root: string; + + beforeEach(async () => { + root = await makeTempRoot(); + // Initialize Agent State (createAgent creates the directory and system_config.yaml). + await createAgent({ root, projectId: PROJECT, agentId: AGENT }); + // Configure Model m1 so recovery doesn't fail on a missing Model (unless the test deletes it on purpose). + await saveProjectConfig(root, PROJECT, { + default_model: { provider: "custom", model_id: "m1" }, + models: [{ provider: "custom", model_id: "m1", context_window: 1000 }], + }); + }); + afterEach(async () => { + await fs.rm(root, { recursive: true, force: true }); + }); + + it("有 Trace 但 Workspace 已删 → 409 session_unrecoverable,保留原文", async () => { + await writeTraceFile(root, PROJECT, AGENT, "2026-07-06", SID, 1, [ + sessionMeta(meta()), // workspace points to a nonexistent directory + userText("hi"), + ]); + const loader = createCoreSessionLoader(root); + const err = await loader.load(row("/tmp/does-not-exist-xyz")).catch((e) => e); + expect(err).toBeInstanceOf(HttpError); + expect((err as HttpError).status).toBe(409); + expect((err as HttpError).code).toBe("session_unrecoverable"); + expect((err as HttpError).message).toContain("Workspace 已不存在"); + }); + + it("有 Trace 但 Model 已从配置移除 → 409 session_unrecoverable", async () => { + const ws = path.join(root, "ws"); + await fs.mkdir(ws, { recursive: true }); + await writeTraceFile(root, PROJECT, AGENT, "2026-07-06", SID, 1, [ + sessionMeta(meta({ model_id: "removed-model", workspace: ws })), + userText("hi"), + ]); + const loader = createCoreSessionLoader(root); + const err = await loader.load(row(ws)).catch((e) => e); + expect(err).toBeInstanceOf(HttpError); + expect((err as HttpError).status).toBe(409); + expect((err as HttpError).code).toBe("session_unrecoverable"); + expect((err as HttpError).message).toContain("Model 不在 Project 配置中"); + }); + + it("无 Trace 且 Workspace 已删(自愈分支)→ 409 workspace_missing", async () => { + const loader = createCoreSessionLoader(root); + const err = await loader.load(row("/tmp/gone-workspace-abc")).catch((e) => e); + expect(err).toBeInstanceOf(HttpError); + expect((err as HttpError).status).toBe(409); + expect((err as HttpError).code).toBe("workspace_missing"); + }); +}); diff --git a/packages/server/test/session-manager.test.ts b/packages/server/test/session-manager.test.ts new file mode 100644 index 0000000..9c32fad --- /dev/null +++ b/packages/server/test/session-manager.test.ts @@ -0,0 +1,550 @@ +/** + * Unit tests for the Session runtime (a fake Session / Loader is injected; no + * real LLM requests are made): driving and state transitions, 409 mutual + * exclusion, the four approval modes and taking effect immediately on change, + * abort collapsing to deny, self-healing id swaps, and LLM / tool errors in the + * message stream being persisted (core doesn't throw, so try/catch can't catch them). + */ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import type { DatabaseSync } from "node:sqlite"; +import { + abortEvent, + approvalDecision, + assistantText, + compactionBegin, + compactionEnd, + requestBegin, + requestEnd, + sessionMeta, + thinkingMessage, + toolCall, + toolCallOutput, + userText, + withOrigin, +} from "@prismshadow/penguin-core"; +import type { ApproveFn, OmniMessage } from "@prismshadow/penguin-core"; +import { openDatabase } from "../src/db/database.js"; +import { HttpError } from "../src/http/errors.js"; +import { SessionsRepo } from "../src/db/repos/sessions.js"; +import type { SessionRow } from "../src/db/repos/sessions.js"; +import { ChannelHub } from "../src/runtime/channel.js"; +import type { ChannelEvent } from "../src/runtime/channel.js"; +import type { ErrorRecordArgs, ErrorSink } from "../src/runtime/error-recorder.js"; +import { SessionManager } from "../src/runtime/session-manager.js"; +import type { RuntimeSession, SessionLoader } from "../src/runtime/session-manager.js"; +import type { TitleRequest } from "../src/runtime/title-generator.js"; +import type { UsageContext } from "../src/runtime/usage-recorder.js"; +import { waitFor } from "./helpers.js"; + +const ROW: SessionRow = { + sessionId: "session-1", + projectId: "p1", + agentId: "a1", + modelId: "m1", + provider: "custom", + workspace: "/tmp/w", + approvalMode: "always-ask", + title: null, + createdAt: "2026-07-06T00:00:00.000Z", +}; + +/** A simple, scriptable fake Session: run yields one tool_call and requests approval for it. */ +function approvalFakeSession(sessionId: string, toolName = "exec_command"): RuntimeSession { + return { + sessionId, + toolPermission: (name) => (name === "read_tool" ? "r" : "rw"), + generateTitle: async () => ({ title: null, usage: null }), + compactability: () => "ok" as const, + async *run(_input: OmniMessage[], opts: { approve: ApproveFn; signal: AbortSignal }) { + const tc = toolCall({ name: toolName, arguments: "{}", toolCallId: "tc-1" }); + yield tc; + const decision = await opts.approve(tc); + yield approvalDecision(decision, "tc-1"); + if (opts.signal.aborted) { + yield abortEvent(); + return; + } + yield assistantText(`decision=${decision}`); + }, + async *compact() { + yield compactionBegin({ reason: "manual", mode: "summarize", context: 1, turns: 1 }); + yield compactionEnd({ reason: "manual", mode: "summarize", status: "completed" }); + }, + }; +} + +describe("session-manager", () => { + let db: DatabaseSync; + let sessions: SessionsRepo; + let channels: ChannelHub; + let recorded: OmniMessage[]; + let recordedCtx: UsageContext[]; + + const makeManager = (loader: SessionLoader, errors?: ErrorSink): SessionManager => + new SessionManager({ + sessions, + channels, + loader, + recorder: { + record: async (ctx, msg) => { + recordedCtx.push(ctx); + recorded.push(msg); + }, + }, + ...(errors ? { errors } : {}), + log: () => {}, + }); + + const loaderOf = (session: RuntimeSession): SessionLoader => ({ load: async () => session }); + + const capture = (sessionId: string): ChannelEvent[] => { + const events: ChannelEvent[] = []; + channels.get(sessionId).subscribe((e) => events.push(e)); + return events; + }; + + const serverEvents = (events: ChannelEvent[]): { type: string; [k: string]: unknown }[] => + events + .filter((e) => e.event === "server_event") + .map((e) => JSON.parse(e.data) as { type: string }); + + beforeEach(() => { + db = openDatabase(":memory:"); + sessions = new SessionsRepo(db); + sessions.insert(ROW); + channels = new ChannelHub(); + recorded = []; + recordedCtx = []; + }); + afterEach(() => { + channels.dispose(); + db.close(); + }); + + it("未知 Session → 404", async () => { + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + const err = await manager.startTask("session-ghost", [userText("x")]).catch((e: unknown) => e); + expect((err as { status: number }).status).toBe(404); + }); + + it("startTask:先 publish 输入,驱动结束置回 idle 并推送 task_state", async () => { + sessions.updateApprovalMode("session-1", "allow-all"); + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + const events = capture("session-1"); + const { sessionId } = await manager.startTask("session-1", [userText("你好")]); + expect(sessionId).toBe("session-1"); + await waitFor(() => manager.statusOf("session-1") === "idle" && recorded.length >= 3); + + // The first entry is the input message (visible to other subscribers), followed by task_state: running. + const first = JSON.parse(events[0]!.data) as { payload: { text: string } }; + expect(first.payload.text).toBe("你好"); + const states = serverEvents(events).filter((e) => e.type === "task_state"); + expect(states.map((s) => s.state)).toEqual(["running", "idle"]); + // Outputs and events are forwarded one by one and handed to the recorder. + expect(recordedCtx[0]).toEqual({ + projectId: "p1", + agentId: "a1", + sessionId: "session-1", + modelId: "m1", + provider: "custom", + }); + }); + + it("消息流里的 LLM / 工具失败经 drive 落库(source=llm / environment,带当前 Session 上下文)", async () => { + sessions.updateApprovalMode("session-1", "allow-all"); + const captured: ErrorRecordArgs[] = []; + // core folds LLM / tool failures into the message stream (no throw): a tool + // failure produces one tool_call_output(failed), an LLM failure produces one + // request_end(failed) + an abort carrying the real reason. + const failing: RuntimeSession = { + sessionId: "session-1", + toolPermission: () => "rw", + generateTitle: async () => ({ title: null, usage: null }), + compactability: () => "ok" as const, + async *run(): AsyncGenerator { + yield requestBegin(); + yield toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc-1" }); + yield toolCallOutput({ + output: "ls: /nope\n[tool error] exit code 2", + toolCallId: "tc-1", + stopReason: "failed", + }); + yield requestEnd("failed"); + yield abortEvent("llm request error: 500 upstream"); + }, + async *compact(): AsyncGenerator {}, + }; + const manager = makeManager(loaderOf(failing), { record: (args) => captured.push(args) }); + await manager.startTask("session-1", [userText("跑")]); + await waitFor(() => manager.statusOf("session-1") === "idle" && captured.length >= 2); + + expect(captured.map((a) => [a.source, a.code, a.kind])).toEqual([ + ["environment", "tool_failed:exec_command", "expected"], // error fed back to the model; the Agent adjusts on its own + ["llm", "llm_failed", "unexpected"], // not retryable, requires human intervention + ]); + expect(captured[0]!.ctx).toEqual({ projectId: "p1", agentId: "a1", sessionId: "session-1" }); + expect(String(captured[0]!.err)).toContain("[tool error] exit code 2"); + expect(String(captured[1]!.err)).toBe("llm request error: 500 upstream"); // the abort's real reason + }); + + it("互斥:运行中再次 startTask → 409 task_in_progress;compact → 409", async () => { + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + await manager.startTask("session-1", [userText("go")]); + await waitFor(() => manager.pendingApprovalCount("session-1") === 1); + + const again = await manager.startTask("session-1", [userText("x")]).catch((e: unknown) => e); + expect((again as { status: number; code: string }).status).toBe(409); + expect((again as { code: string }).code).toBe("task_in_progress"); + const compact = await manager.startCompact("session-1").catch((e: unknown) => e); + expect((compact as { status: number }).status).toBe(409); + + manager.decideApproval("session-1", "tc-1", "allow"); + await waitFor(() => manager.statusOf("session-1") === "idle"); + }); + + it("always-ask:登记未决审批并推送 approval_request;决定后继续", async () => { + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + const events = capture("session-1"); + await manager.startTask("session-1", [userText("go")]); + await waitFor(() => manager.pendingApprovalCount("session-1") === 1); + + const requests = serverEvents(events).filter((e) => e.type === "approval_request"); + expect(requests).toHaveLength(1); + expect( + (requests[0]!.toolCall as { payload: { tool_call_id: string } }).payload.tool_call_id, + ).toBe("tc-1"); + expect(manager.pendingApprovals("session-1")).toHaveLength(1); + + expect(manager.decideApproval("session-1", "tc-404", "allow")).toBe(false); + expect(manager.decideApproval("session-1", "tc-1", "allow")).toBe(true); + await waitFor(() => manager.statusOf("session-1") === "idle"); + expect(manager.pendingApprovalCount("session-1")).toBe(0); + const texts = recorded + .filter((m) => (m.payload as { type?: string }).type === "text") + .map((m) => (m.payload as { text: string }).text); + expect(texts).toContain("decision=allow"); + }); + + it("deny-all / read-only 自动判定(不转人工)", async () => { + sessions.updateApprovalMode("session-1", "deny-all"); + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + await manager.startTask("session-1", [userText("go")]); + await waitFor(() => manager.statusOf("session-1") === "idle"); + expect( + recorded.some( + (m) => + (m.payload as { type?: string; decision?: string }).type === "approval_decision" && + (m.payload as { decision: string }).decision === "deny", + ), + ).toBe(true); + + // read-only: read-only tools are auto-approved. + recorded = []; + sessions.updateApprovalMode("session-1", "read-only"); + const manager2 = makeManager(loaderOf(approvalFakeSession("session-1", "read_tool"))); + await manager2.startTask("session-1", [userText("go")]); + await waitFor(() => manager2.statusOf("session-1") === "idle"); + expect(recorded.some((m) => (m.payload as { text?: string }).text === "decision=allow")).toBe( + true, + ); + }); + + it("子会话(origin)登记:session_meta 落库,标题交模型据子会话自己的对话生成", async () => { + const fake: RuntimeSession = { + sessionId: "session-1", + toolPermission: () => "rw", + generateTitle: async () => ({ title: null, usage: null }), + compactability: () => "ok" as const, + async *run() { + // The parent-level run_subagent call (no origin): its prompt becomes the sub-session title. + yield toolCall({ + name: "run_subagent", + arguments: JSON.stringify({ prompt: "研究一下这个问题的背景资料" }), + toolCallId: "sub-1", + }); + const hop = "child-1"; + yield withOrigin( + sessionMeta({ + session_id: "child-1", + model_id: "m-child", + provider: "custom", + model_context_window: 1000, + system_prompt: "sys", + tools: [], + thinking_level: "default", + agent_state: "/root/p1/child_agent/agent_state", + workspace: "/tmp/w-child", + }), + hop, + ); + yield withOrigin(assistantText("child done"), hop); + yield assistantText("done"); + }, + async *compact() {}, + }; + const notified: Array<{ ctx: UsageContext; req: TitleRequest }> = []; + const manager = new SessionManager({ + sessions, + channels, + loader: loaderOf(fake), + recorder: { record: async () => {} }, + titles: { + maybeGenerate: (ctx, _session, req) => notified.push({ ctx, req }), + }, + log: () => {}, + }); + await manager.startTask("session-1", [userText("go")]); + await waitFor(() => manager.statusOf("session-1") === "idle"); + const child = sessions.findById("child-1"); + expect(child).not.toBeNull(); + expect(child?.agentId).toBe("child_agent"); + expect(child?.modelId).toBe("m-child"); + expect(child?.workspace).toBe("/tmp/w-child"); + // Title left blank: produced by the title generator from the sub-session's own conversation (falls back to the prompt's first line on failure). + expect(child?.title).toBeNull(); + + await waitFor(() => notified.length === 2); + const childTitle = notified.find((n) => n.ctx.sessionId === "child-1"); + expect(childTitle).toBeTruthy(); + // Explicit material override for the sub-session: user material = the prompt that spawned it; assistant material = the sub-session's **own** model output. + expect(childTitle!.req.material).toEqual({ + userText: "研究一下这个问题的背景资料", + assistantText: "child done", + }); + expect(childTitle!.req.fallbackText).toBe("研究一下这个问题的背景资料"); + // Session/Agent record the sub-session, but modelId records the parent — the request runs on the parent's bare LLM. + expect(childTitle!.ctx).toMatchObject({ agentId: "child_agent", modelId: "m1" }); + // The sub-session has no SSE channel of its own: title events are delivered over the parent session's channel. + expect(childTitle!.req.notifyOn).toBe("session-1"); + }); + + it("审批模式即改即生效:运行中 PATCH 后下一次决策用新模式", async () => { + // A fake Session that requests approval twice. + const fake: RuntimeSession = { + sessionId: "session-1", + toolPermission: () => "rw", + generateTitle: async () => ({ title: null, usage: null }), + compactability: () => "ok" as const, + async *run(_input, opts) { + const tc1 = toolCall({ name: "t1", arguments: "{}", toolCallId: "tc-1" }); + yield tc1; + yield approvalDecision(await opts.approve(tc1), "tc-1"); + const tc2 = toolCall({ name: "t2", arguments: "{}", toolCallId: "tc-2" }); + yield tc2; + yield approvalDecision(await opts.approve(tc2), "tc-2"); + }, + async *compact() {}, + }; + const manager = makeManager(loaderOf(fake)); + await manager.startTask("session-1", [userText("go")]); + await waitFor(() => manager.pendingApprovalCount("session-1") === 1); + // Switch to allow-all while the first request is pending human review: the second no longer needs one. + sessions.updateApprovalMode("session-1", "allow-all"); + manager.decideApproval("session-1", "tc-1", "deny"); + await waitFor(() => manager.statusOf("session-1") === "idle"); + const decisions = recorded + .filter((m) => (m.payload as { type?: string }).type === "approval_decision") + .map((m) => (m.payload as { decision: string }).decision); + expect(decisions).toEqual(["deny", "allow"]); + expect(manager.pendingApprovalCount("session-1")).toBe(0); + }); + + it("abort:未决审批收敛为 deny 再触发 AbortSignal", async () => { + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + expect(manager.abortTask("session-1")).toBe(false); // no Task in progress → no-op + await manager.startTask("session-1", [userText("go")]); + await waitFor(() => manager.pendingApprovalCount("session-1") === 1); + expect(manager.abortTask("session-1")).toBe(true); + await waitFor(() => manager.statusOf("session-1") === "idle"); + const payloads = recorded.map((m) => m.payload as { type?: string; decision?: string }); + expect(payloads.some((p) => p.type === "approval_decision" && p.decision === "deny")).toBe( + true, + ); + expect(payloads.some((p) => p.type === "abort")).toBe(true); + }); + + it("beginSessionDeletion:中断活跃运行、清出活跃表并标记删除中(新任务 409),end 后恢复", async () => { + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + expect(manager.beginSessionDeletion("session-1")).toEqual([]); // no active entry + // While deletion is in progress: a new Task is rejected with 409 (prevents resurrection). + const rejected = await manager.startTask("session-1", [userText("x")]).catch((e: unknown) => e); + expect((rejected as { status?: number }).status).toBe(409); + manager.endSessionDeletion("session-1"); + // Runs normally once the deletion flag is cleared. + await manager.startTask("session-1", [userText("go")]); + await waitFor(() => manager.pendingApprovalCount("session-1") === 1); + const runnings = manager.beginSessionDeletion("session-1"); + expect(runnings.length).toBe(1); + await Promise.allSettled(runnings); + expect(manager.statusOf("session-1")).toBe("idle"); // entry has been removed + manager.endSessionDeletion("session-1"); + }); + + it("自愈:loader 返回新 session_id 时更新索引主键并返回当前实际 id", async () => { + sessions.updateApprovalMode("session-1", "allow-all"); + const manager = makeManager(loaderOf(approvalFakeSession("session-2-healed"))); + const { sessionId } = await manager.startTask("session-1", [userText("go")]); + expect(sessionId).toBe("session-2-healed"); + expect(sessions.findById("session-1")).toBeNull(); + expect(sessions.findById("session-2-healed")).not.toBeNull(); + await waitFor(() => manager.statusOf("session-2-healed") === "idle"); + // Usage is attributed under the new id. + expect(recordedCtx[0]!.sessionId).toBe("session-2-healed"); + }); + + it("通道被回收重建后 drive 仍发到当前通道(每次 publish 前重新 get)", async () => { + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + await manager.startTask("session-1", [userText("go")]); + await waitFor(() => manager.pendingApprovalCount("session-1") === 1); + + // No subscribers, no publish while waiting for approval: simulate idle sweeping removing the old channel. + channels.sweep(Date.now() + 60 * 60 * 1000); + // Reconnect: hub.get creates a brand-new channel, and we subscribe to it. + const events = capture("session-1"); + manager.decideApproval("session-1", "tc-1", "allow"); + await waitFor(() => manager.statusOf("session-1") === "idle"); + + // The remaining output and task_state after approval must land on the new channel (the old reference should already be stale). + const texts = events + .filter((e) => e.event === undefined) + .map((e) => (JSON.parse(e.data) as { payload: { type?: string; text?: string } }).payload) + .filter((p) => p.type === "text") + .map((p) => p.text); + expect(texts).toContain("decision=allow"); + const states = serverEvents(events).filter((e) => e.type === "task_state"); + expect(states.map((s) => s.state)).toContain("idle"); + }); + + it("abortProject:返回进行中的驱动 Promise,等待后收尾完成", async () => { + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + expect(manager.abortProject("p1")).toEqual([]); // no active runs → empty array + await manager.startTask("session-1", [userText("go")]); + await waitFor(() => manager.pendingApprovalCount("session-1") === 1); + + const runnings = manager.abortProject("p1"); + expect(runnings).toHaveLength(1); + await Promise.allSettled(runnings); + // Wrap-up complete: the abort cleanup (abort event) has been written, and the entry has been removed from the active table. + expect(recorded.some((m) => (m.payload as { type?: string }).type === "abort")).toBe(true); + expect(manager.statusOf("session-1")).toBe("idle"); + }); + + it("loader 抛 HttpError(如 workspace_missing 409)→ 原样透传,不被重复包装", async () => { + const loader: SessionLoader = { + load: async () => { + throw new HttpError(409, "workspace_missing", "Workspace 已不存在。"); + }, + }; + const manager = makeManager(loader); + const err = await manager.startTask("session-1", [userText("x")]).catch((e: unknown) => e); + expect((err as { status: number; code: string }).status).toBe(409); + expect((err as { code: string }).code).toBe("workspace_missing"); + }); + + it("shutdown 置位后拒收新任务(503 shutting_down)", async () => { + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + await manager.shutdown(); + const err = await manager.startTask("session-1", [userText("x")]).catch((e: unknown) => e); + expect((err as { status: number; code: string }).status).toBe(503); + expect((err as { code: string }).code).toBe("shutting_down"); + const compactErr = await manager.startCompact("session-1").catch((e: unknown) => e); + expect((compactErr as { status: number }).status).toBe(503); + }); + + it("sweepIdle:空闲超时的 entry 被淘汰(下次访问经 loader 重新装载)", async () => { + let loads = 0; + const loader: SessionLoader = { + load: async () => { + loads++; + return approvalFakeSession("session-1"); + }, + }; + const manager = makeManager(loader); + manager.adopt(ROW, approvalFakeSession("session-1")); + // Not evicted before timeout: startTask reuses the active-table entry, bypassing the loader. + manager.sweepIdle(Date.now() + 1000, 30 * 60 * 1000); + sessions.updateApprovalMode("session-1", "allow-all"); + await manager.startTask("session-1", [userText("a")]); + await waitFor(() => manager.statusOf("session-1") === "idle"); + expect(loads).toBe(0); + + // Evicted after timeout: once the entry is released, the next startTask reloads it. + manager.sweepIdle(Date.now() + 31 * 60 * 1000, 30 * 60 * 1000); + await manager.startTask("session-1", [userText("b")]); + await waitFor(() => manager.statusOf("session-1") === "idle"); + expect(loads).toBe(1); + }); + + it("sweepIdle:运行中 / 有未决审批的 entry 不淘汰", async () => { + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + await manager.startTask("session-1", [userText("go")]); + await waitFor(() => manager.pendingApprovalCount("session-1") === 1); + manager.sweepIdle(Date.now() + 24 * 60 * 60 * 1000, 30 * 60 * 1000); + // The entry is still there: the approval can be decided and wraps up normally. + expect(manager.statusOf("session-1")).toBe("running"); + expect(manager.decideApproval("session-1", "tc-1", "allow")).toBe(true); + await waitFor(() => manager.statusOf("session-1") === "idle"); + }); + + it("compact:置 compacting、输出进通道、结束回 idle", async () => { + const manager = makeManager(loaderOf(approvalFakeSession("session-1"))); + const events = capture("session-1"); + await manager.startCompact("session-1"); + await waitFor(() => manager.statusOf("session-1") === "idle" && recorded.length >= 2); + const states = serverEvents(events).filter((e) => e.type === "task_state"); + expect(states.map((s) => s.state)).toEqual(["compacting", "idle"]); + expect(recorded.map((m) => (m.payload as { type: string }).type)).toEqual([ + "compaction_begin", + "compaction_end", + ]); + }); + + it("Task 完成后通知标题生成:兜底素材取用户 text、生成素材由 Session 自采;压缩不通知", async () => { + const notified: { ctx: UsageContext; session: unknown; req: TitleRequest }[] = []; + const plainSession: RuntimeSession = { + sessionId: "session-1", + toolPermission: () => "rw", + generateTitle: async () => ({ title: null, usage: null }), + compactability: () => "ok" as const, + async *run() { + yield thinkingMessage("思考中"); + yield assistantText("答案A"); + yield withOrigin(assistantText("子会话文本"), "session-sub"); + yield assistantText("答案B"); + }, + async *compact() { + yield compactionBegin({ reason: "manual", mode: "summarize", context: 1, turns: 1 }); + yield compactionEnd({ reason: "manual", mode: "summarize", status: "completed" }); + }, + }; + const manager = new SessionManager({ + sessions, + channels, + loader: loaderOf(plainSession), + recorder: { record: async () => {} }, + titles: { + maybeGenerate: (ctx, session, req) => notified.push({ ctx, session, req }), + }, + log: () => {}, + }); + + await manager.startTask("session-1", [userText("问题1"), userText("问题2")]); + await waitFor(() => notified.length === 1); + expect(notified[0]!.req.fallbackText).toBe("问题1\n问题2"); + // No material override for the main session: material is gathered by the core Session during run. + expect(notified[0]!.req.material).toBeUndefined(); + expect(notified[0]!.session).toBe(plainSession); + expect(notified[0]!.ctx).toMatchObject({ + projectId: "p1", + agentId: "a1", + sessionId: "session-1", + modelId: "m1", + provider: "custom", + }); + + await manager.startCompact("session-1"); + await waitFor(() => manager.statusOf("session-1") === "idle"); + await new Promise((r) => setTimeout(r, 10)); + expect(notified.length).toBe(1); + }); +}); diff --git a/packages/server/test/skills.test.ts b/packages/server/test/skills.test.ts new file mode 100644 index 0000000..b154cd0 --- /dev/null +++ b/packages/server/test/skills.test.ts @@ -0,0 +1,211 @@ +/** + * Integration tests for the Skill routes: library catalog structure (any logged-in user), member + * install/uninstall with 404 for outsiders, 404 for unknown skills, installed + * files matching the library content, idempotent update on reinstall, the + * directory disappearing after uninstall, and default_agent starting with all + * skills installed while a newly created plain Agent has none. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { skillsDir } from "@prismshadow/penguin-core"; +import { librarySkill, loadLibrarySkills } from "@prismshadow/penguin-skills"; +import type { + AgentSkillsResponse, + ProjectCreateResponse, + SkillLibraryResponse, +} from "../src/api/types.js"; +import { apiClient, createTestApp, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +describe("skills api", () => { + let t: TestApp; + let owner: ReturnType; + let member: ReturnType; + let outsider: ReturnType; + let projectId: string; + const base = (agentId: string) => `/api/projects/${projectId}/agents/${agentId}/skills`; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner_s"); + const b = await provisionUser(t.app, "member_s"); + const c = await provisionUser(t.app, "outsider_s"); + owner = apiClient(t.app, a.cookie); + member = apiClient(t.app, b.cookie); + outsider = apiClient(t.app, c.cookie); + const created = (await ( + await owner.post("/api/projects", { projectId: "owner_s-skills", name: "技能项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + expect( + (await owner.post(`/api/projects/${projectId}/members`, { userId: "member_s" })).status, + ).toBe(201); + }); + afterEach(async () => { + await t.cleanup(); + }); + + /** Creates a plain Agent with no Skills preinstalled. */ + async function createPlainAgent(agentId: string): Promise { + const res = await owner.post(`/api/projects/${projectId}/agents`, { agentId }); + expect(res.status).toBe(201); + } + + it("GET /api/skills:分组与 metadata、短描述、图标齐全且不下发正文", async () => { + const res = await member.get("/api/skills"); + expect(res.status).toBe(200); + const body = (await res.json()) as SkillLibraryResponse; + expect(body.groups.map((g) => g.id)).toEqual([ + "agent-development", + "data-analysis", + "penguin-development", + "web-development", + "software-engineering", + ]); + for (const group of body.groups) { + expect(group.title.length).toBeGreaterThan(0); + // The Chinese group title is passed through from the skills package (the UI + // picks a language); groups no longer carry a description. + expect(group.titleZh).toBeTruthy(); + expect("description" in group).toBe(false); + } + // Members within a group follow the SKILL_GROUPS list order (as ungrouped by loadSkillGroups). + expect(body.groups[0]!.skills.map((s) => s.name)).toEqual([ + "agent-creation", + "benchmark-design", + "agent-evaluation", + "agent-optimization", + ]); + expect(body.groups[1]!.skills.map((s) => s.name)).toEqual(["data-analysis"]); + expect(body.groups[2]!.skills.map((s) => s.name)).toEqual([ + "penguin-sdk", + "penguin-cli", + "agenthub-models", + ]); + expect(body.groups[3]!.skills.map((s) => s.name)).toEqual(["web-design"]); + expect(body.groups[4]!.skills.map((s) => s.name)).toEqual(["software-engineering"]); + const skills = body.groups.flatMap((g) => g.skills); + for (const skill of skills) { + expect(skill.name.length).toBeGreaterThan(0); + expect(skill.description.length).toBeGreaterThan(0); + // The short description (preferred in compact spots like cards) and custom + // icon (raw icon.svg) are passed through conditionally for every returned skill. + expect(skill.shortDescription, skill.name).toBeTruthy(); + expect(skill.shortDescriptionZh, skill.name).toBeTruthy(); + expect(skill.icon, skill.name).toContain(" { + await createPlainAgent("bare_agent"); + const url = base("bare_agent"); + + // Member installs two Skills: 201 returns the updated list (sorted by name). + const res = await member.post(url, { names: ["penguin-sdk", "agent-creation"] }); + expect(res.status).toBe(201); + const body = (await res.json()) as AgentSkillsResponse; + expect(body.skills.map((s) => s.name)).toEqual(["agent-creation", "penguin-sdk"]); + // The installed list likewise passes through the short description and icon + // (icon.svg is copied on install, identical to the library's original). + const installed = body.skills.find((s) => s.name === "penguin-sdk")!; + expect(installed.shortDescription).toBeTruthy(); + expect(installed.icon).toBe(librarySkill("penguin-sdk")!.icon); + + // The on-disk content matches the library's SKILL.md verbatim (including + // frontmatter), and icon.svg is written alongside it. + const skillFile = (name: string) => + path.join(skillsDir(t.root, projectId, "bare_agent"), name, "SKILL.md"); + expect(await fs.readFile(skillFile("penguin-sdk"), "utf8")).toBe( + librarySkill("penguin-sdk")!.content, + ); + expect( + await fs.readFile( + path.join(skillsDir(t.root, projectId, "bare_agent"), "penguin-sdk", "icon.svg"), + "utf8", + ), + ).toBe(librarySkill("penguin-sdk")!.icon); + + // Member uninstalls: 204, the whole skills// directory disappears, and the list is updated. + expect((await member.delete(`${url}/penguin-sdk`)).status).toBe(204); + await expect(fs.access(path.dirname(skillFile("penguin-sdk")))).rejects.toThrow(); + const after = (await (await member.get(url)).json()) as AgentSkillsResponse; + expect(after.skills.map((s) => s.name)).toEqual(["agent-creation"]); + + // Deleting a Skill that isn't installed (or was already uninstalled) → 404. + expect((await member.delete(`${url}/penguin-sdk`)).status).toBe(404); + }); + + it("重复安装幂等更新:手改落盘内容后再装,恢复为库内容", async () => { + await createPlainAgent("update_agent"); + const url = base("update_agent"); + expect((await owner.post(url, { names: ["penguin-cli"] })).status).toBe(201); + + // Simulate stale/tampered on-disk content. + const file = path.join(skillsDir(t.root, projectId, "update_agent"), "penguin-cli", "SKILL.md"); + await fs.writeFile(file, "---\nname: penguin-cli\nversion: 0\n---\nstale\n", "utf8"); + + const res = await owner.post(url, { names: ["penguin-cli"] }); + expect(res.status).toBe(201); + const body = (await res.json()) as AgentSkillsResponse; + expect(body.skills.map((s) => s.name)).toEqual(["penguin-cli"]); + expect(await fs.readFile(file, "utf8")).toBe(librarySkill("penguin-cli")!.content); + }); + + it("未知技能 404 unknown_skill,且不产生半安装状态", async () => { + await createPlainAgent("strict_agent"); + const url = base("strict_agent"); + const res = await owner.post(url, { names: ["penguin-sdk", "no-such-skill"] }); + expect(res.status).toBe(404); + const err = (await res.json()) as { error: { code: string; message: string } }; + expect(err.error.code).toBe("unknown_skill"); + expect(err.error.message).toContain("no-such-skill"); + // Whole request rejected: even the valid library skill was not written to disk. + const list = (await (await owner.get(url)).json()) as AgentSkillsResponse; + expect(list.skills).toEqual([]); + }); + + it("请求体校验 400:names 缺失/空数组/含非字符串", async () => { + await createPlainAgent("valid_agent"); + const url = base("valid_agent"); + for (const body of [{}, { names: [] }, { names: ["penguin-sdk", 1] }, { names: [""] }]) { + expect((await owner.post(url, body)).status, JSON.stringify(body)).toBe(400); + } + }); + + it("外人一律 404(读取、安装、卸载);Agent 不存在 404", async () => { + const url = base("default_agent"); + expect((await outsider.get(url)).status).toBe(404); + expect((await outsider.post(url, { names: ["penguin-sdk"] })).status).toBe(404); + expect((await outsider.delete(`${url}/penguin-sdk`)).status).toBe(404); + // The library catalog isn't scoped under a Project prefix: any logged-in user can read it. + expect((await outsider.get("/api/skills")).status).toBe(200); + // Agent doesn't exist: even a member gets 404. + expect((await member.get(base("no_such_agent"))).status).toBe(404); + }); + + it("default_agent 初始即装库内全部技能;新建普通 Agent 技能为空", async () => { + const res = await member.get(base("default_agent")); + expect(res.status).toBe(200); + const body = (await res.json()) as AgentSkillsResponse; + // loadLibrarySkills itself sorts by name, matching the installed-list ordering. + expect(body.skills.map((s) => s.name)).toEqual(loadLibrarySkills().map((s) => s.name)); + // The installed list likewise passes through the Chinese description and the + // short description/icon (listInstalledSkills parses these from the on-disk + // frontmatter and icon.svg). + for (const skill of body.skills) { + expect(skill.shortDescription, skill.name).toBeTruthy(); + expect(skill.shortDescriptionZh, skill.name).toBeTruthy(); + expect(skill.icon, skill.name).toContain(" { + expect(res.status).toBe(200); + const reader = res.body!.getReader(); + const decoder = new TextDecoder(); + const frames: SseFrame[] = []; + let buf = ""; + const deadline = Date.now() + timeoutMs; + try { + while (frames.length < count) { + if (Date.now() > deadline) throw new Error(`SSE 读取超时(已有 ${frames.length} 帧)`); + const { done, value } = await reader.read(); + if (done) break; + buf += decoder.decode(value, { stream: true }); + let idx: number; + while ((idx = buf.indexOf("\n\n")) !== -1) { + const raw = buf.slice(0, idx); + buf = buf.slice(idx + 2); + if (raw.startsWith(":") || raw.trim() === "") continue; // heartbeat/empty frame + const frame: SseFrame = { data: "" }; + for (const line of raw.split("\n")) { + if (line.startsWith("event:")) frame.event = line.slice(6).trim(); + else if (line.startsWith("id:")) frame.id = line.slice(3).trim(); + else if (line.startsWith("data:")) frame.data += line.slice(5).trim(); + } + frames.push(frame); + } + } + } finally { + await reader.cancel().catch(() => {}); + } + return frames; +} + +/** Fake Session that requests one approval (for the running state and approval-replay scenarios). */ +function approvalFakeSession(sessionId: string): RuntimeSession { + return { + sessionId, + toolPermission: () => "rw", + generateTitle: async () => ({ title: null, usage: null }), + compactability: () => "ok" as const, + async *run(_input: OmniMessage[], opts: { approve: ApproveFn; signal: AbortSignal }) { + const tc = toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc-sse" }); + yield tc; + const decision = await opts.approve(tc); + yield approvalDecision(decision, "tc-sse"); + yield assistantText("done"); + }, + async *compact() {}, + }; +} + +describe("sse-stream", () => { + let t: TestApp; + let cookie: string; + let row: SessionRow; + + beforeEach(async () => { + t = await createTestApp(); + ({ cookie } = await provisionUser(t.app, "streamer")); + row = { + sessionId: SID, + // streamer's own initial Project (default_project belongs to admin; others + // get 404 via the index lookup). + projectId: "streamer-default_project", + agentId: "default_agent", + modelId: "m1", + provider: "custom", + workspace: "/tmp/w", + approvalMode: "always-ask", + title: null, + createdAt: new Date().toISOString(), + }; + t.deps.sessionsRepo.insert(row); + }); + afterEach(async () => { + await t.cleanup(); + }); + + const getStream = (headers: Record = {}) => + t.app.request(`/api/sessions/${SID}/stream`, { headers: { cookie, ...headers } }); + + it("FD-1:新订阅第一条恒为 task_state 快照(idle)", async () => { + const frames = await readSseFrames(await getStream(), 1); + expect(frames[0]!.event).toBe("server_event"); + expect(JSON.parse(frames[0]!.data)).toEqual({ type: "task_state", state: "idle" }); + expect(frames[0]!.id).toMatch(/^[0-9a-f]{8}-\d+$/); // FD-2: opaque string id + }); + + it("FD-1:运行中订阅收到 task_state: running,再补发未决审批", async () => { + t.deps.manager.adopt(row, approvalFakeSession(SID)); + await t.deps.manager.startTask(SID, [userText("go")]); + await waitFor(() => t.deps.manager.pendingApprovalCount(SID) === 1); + + const frames = await readSseFrames(await getStream(), 2); + expect(JSON.parse(frames[0]!.data)).toEqual({ type: "task_state", state: "running" }); + const approval = JSON.parse(frames[1]!.data) as { + type: string; + toolCall: { payload: { tool_call_id: string } }; + }; + expect(approval.type).toBe("approval_request"); + expect(approval.toolCall.payload.tool_call_id).toBe("tc-sse"); + + t.deps.manager.abortTask(SID); + await waitFor(() => t.deps.manager.statusOf(SID) === "idle"); + }); + + it("FD-2:Last-Event-ID 纪元不一致 → 先发 resync_required 再发 task_state 快照", async () => { + t.deps.channels.get(SID).publish(userText("旧事件")); + const frames = await readSseFrames( + await getStream({ "Last-Event-ID": "deadbeef-1" }), // guaranteed to differ from the current channel epoch + 2, + ); + expect(JSON.parse(frames[0]!.data)).toEqual({ type: "resync_required" }); + expect(JSON.parse(frames[1]!.data)).toEqual({ type: "task_state", state: "idle" }); + }); + + it("FD-2:同纪元 Last-Event-ID 命中缓冲 → 补发其后事件,再发 task_state 快照", async () => { + const channel = t.deps.channels.get(SID); + const first = channel.publish(userText("m1")); + channel.publish(userText("m2")); + const frames = await readSseFrames(await getStream({ "Last-Event-ID": first.id }), 2); + expect(frames[0]!.event).toBeUndefined(); // replayed OmniMessage + expect((JSON.parse(frames[0]!.data) as { payload: { text: string } }).payload.text).toBe("m2"); + expect(JSON.parse(frames[1]!.data)).toEqual({ type: "task_state", state: "idle" }); + }); +}); diff --git a/packages/server/test/title-generator.test.ts b/packages/server/test/title-generator.test.ts new file mode 100644 index 0000000..cf615aa --- /dev/null +++ b/packages/server/test/title-generator.test.ts @@ -0,0 +1,184 @@ +/** + * Unit tests for the Session title policy layer (the generation logic lives in + * core `session.generateTitle`; a fake implementation is injected here): + * persisting + event push + usage accounted as token usage, idempotency (no + * regeneration when a title already exists / generation is in flight), and + * silent failure. + */ +import { describe, it, expect, beforeEach, afterEach } from "vitest"; +import type { DatabaseSync } from "node:sqlite"; +import type { OmniMessage, SessionTitleResult } from "@prismshadow/penguin-core"; +import { openDatabase } from "../src/db/database.js"; +import { SessionsRepo } from "../src/db/repos/sessions.js"; +import type { SessionRow } from "../src/db/repos/sessions.js"; +import { ChannelHub } from "../src/runtime/channel.js"; +import type { ChannelEvent } from "../src/runtime/channel.js"; +import { TitleGenerator } from "../src/runtime/title-generator.js"; +import type { UsageContext } from "../src/runtime/usage-recorder.js"; +import { waitFor } from "./helpers.js"; + +const ROW: SessionRow = { + sessionId: "session-t1", + projectId: "p1", + agentId: "a1", + modelId: "m1", + provider: "custom", + workspace: "/tmp/w", + approvalMode: "always-ask", + title: null, + createdAt: "2026-07-07T00:00:00.000Z", +}; + +const CTX: UsageContext = { + projectId: "p1", + agentId: "a1", + sessionId: "session-t1", + modelId: "m1", + provider: "custom", +}; + +/** Fake session: generateTitle returns the given result and records the call count/arguments. */ +function fakeSession(result: SessionTitleResult, calls: { count: number; args: unknown[] }) { + return { + generateTitle: async (args: unknown) => { + calls.count += 1; + calls.args.push(args); + return result; + }, + }; +} + +describe("title-generator", () => { + let db: DatabaseSync; + let sessions: SessionsRepo; + let channels: ChannelHub; + let recorded: OmniMessage[]; + + const makeGenerator = (): TitleGenerator => + new TitleGenerator({ + sessions, + channels, + recorder: { + record: async (_ctx, msg) => { + recorded.push(msg); + }, + }, + log: () => {}, + }); + + const captureChannel = (): ChannelEvent[] => { + const events: ChannelEvent[] = []; + channels.get(ROW.sessionId).subscribe((e) => events.push(e)); + return events; + }; + + const serverEvents = (events: ChannelEvent[]): { type: string; [k: string]: unknown }[] => + events + .filter((e) => e.event === "server_event") + .map((e) => JSON.parse(e.data) as { type: string }); + + beforeEach(() => { + db = openDatabase(":memory:"); + sessions = new SessionsRepo(db); + sessions.insert(ROW); + channels = new ChannelHub(); + recorded = []; + }); + afterEach(() => { + channels.dispose(); + db.close(); + }); + + it("生成标题:落库 + 推送 session_title + 用量折算入账;素材缺省由 Session 自采", async () => { + const events = captureChannel(); + const calls = { count: 0, args: [] as unknown[] }; + const gen = makeGenerator(); + gen.maybeGenerate( + CTX, + fakeSession( + { + title: "Tailwind 主题配置", + usage: { cache_read: 1, cache_write: 2, output: 3, total: 6 }, + }, + calls, + ), + { fallbackText: "解释一下 @theme" }, + ); + await waitFor(() => sessions.findById(ROW.sessionId)?.title !== null); + + expect(sessions.findById(ROW.sessionId)?.title).toBe("Tailwind 主题配置"); + // No material override passed: generateTitle is called with no argument, and the core Session gathers its own material. + expect(calls.args[0]).toBeUndefined(); + expect( + serverEvents(events).some( + (e) => e.type === "session_title" && e.title === "Tailwind 主题配置", + ), + ).toBe(true); + // usage is converted into token_usage and handed to the recorder (metered normally, same as a real call). + const usageMsg = recorded.find((m) => (m.payload as { type?: string }).type === "token_usage"); + expect((usageMsg?.payload as { request?: { total: number } }).request?.total).toBe(6); + }); + + it("素材覆盖(子会话场景)原样传给 generateTitle", async () => { + const calls = { count: 0, args: [] as unknown[] }; + const gen = makeGenerator(); + const material = { userText: "子会话 prompt", assistantText: "子会话回答" }; + gen.maybeGenerate(CTX, fakeSession({ title: "子标题", usage: null }, calls), { + fallbackText: "子会话 prompt", + material, + }); + await waitFor(() => sessions.findById(ROW.sessionId)?.title !== null); + expect(calls.args[0]).toEqual({ material }); + expect(sessions.findById(ROW.sessionId)?.title).toBe("子标题"); + }); + + it("已有标题不再生成(不发起单发请求)", async () => { + sessions.updateTitle(ROW.sessionId, "已有标题"); + const calls = { count: 0, args: [] as unknown[] }; + makeGenerator().maybeGenerate(CTX, fakeSession({ title: "新标题", usage: null }, calls), { + fallbackText: "u", + }); + await new Promise((r) => setTimeout(r, 20)); + expect(calls.count).toBe(0); + expect(sessions.findById(ROW.sessionId)?.title).toBe("已有标题"); + }); + + it("LLM 返回 null(请求失败/空结果)时用兜底素材首行落库;usage 仍入账", async () => { + const calls = { count: 0, args: [] as unknown[] }; + const gen = makeGenerator(); + gen.maybeGenerate( + CTX, + fakeSession( + { title: null, usage: { cache_read: 0, cache_write: 0, output: 1, total: 1 } }, + calls, + ), + { fallbackText: "配置 Tailwind 主题\n第二行" }, + ); + // Fallback = the material's first non-empty line, sanitized and truncated, guaranteeing a title is always produced. + await waitFor(() => sessions.findById(ROW.sessionId)?.title !== null); + expect(sessions.findById(ROW.sessionId)?.title).toBe("配置 Tailwind 主题"); + // The one-off request's usage is still recorded normally. + await waitFor(() => recorded.length > 0); + }); + + it("LLM 返回 null 且兜底素材为空 → 标题保持 NULL(下次可重试)", async () => { + const calls = { count: 0, args: [] as unknown[] }; + const gen = makeGenerator(); + gen.maybeGenerate(CTX, fakeSession({ title: null, usage: null }, calls), { + fallbackText: " ", + }); + await waitFor(() => calls.count >= 1); + expect(sessions.findById(ROW.sessionId)?.title).toBeNull(); + }); + + it("LLM 返回 null 且兜底素材为纯标点 → 兜底退回截断原文(不落 NULL)", async () => { + const calls = { count: 0, args: [] as unknown[] }; + const gen = makeGenerator(); + // sanitizeTitle strips "???" down to empty — the fallback must revert to the truncated original text so a title is still produced. + gen.maybeGenerate(CTX, fakeSession({ title: null, usage: null }, calls), { + fallbackText: "???", + }); + await waitFor(() => sessions.findById(ROW.sessionId)?.title !== null); + expect(sessions.findById(ROW.sessionId)?.title).toBe("???"); + }); +}); diff --git a/packages/server/test/trace-service.test.ts b/packages/server/test/trace-service.test.ts new file mode 100644 index 0000000..22c4840 --- /dev/null +++ b/packages/server/test/trace-service.test.ts @@ -0,0 +1,675 @@ +/** + * Unit tests for the Trace service: multi-file history concatenation, file + * listing, pagination, performance-analysis derivation, and Agent-level + * drill-down browsing. + */ +import fs from "node:fs/promises"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { + abortEvent, + approvalDecision, + assistantText, + compactionBegin, + compactionEnd, + imageUrlMessage, + requestBegin, + requestEnd, + sessionMeta, + thinkingMessage, + tokenUsage, + toolCall, + toolCallOutput, + userText, +} from "@prismshadow/penguin-core"; +import type { OmniMessage, SessionMetaPayload, TokenCounts } from "@prismshadow/penguin-core"; +import { TraceService } from "../src/services/trace-service.js"; +import { makeTempRoot, writeTraceFile } from "./helpers.js"; + +const P = "project-t"; +const A = "agent-t"; +const S = "session-2026-07-05-10-00-00-aabbccdd"; + +function at(ts: string, msg: OmniMessage): OmniMessage { + return { ...msg, timestamp: ts }; +} + +function counts(total: number): TokenCounts { + return { cache_read: 0, cache_write: 0, output: 0, total }; +} + +/** Request usage with real three-bucket counts (both the context snapshot and the TPS numerator are derived from this). */ +function buckets(cacheRead: number, cacheWrite: number, output: number): TokenCounts { + return { + cache_read: cacheRead, + cache_write: cacheWrite, + output, + total: cacheRead + cacheWrite + output, + }; +} + +function metaPayload(): SessionMetaPayload { + return { + session_id: S, + model_id: "m1", + provider: "custom", + model_context_window: 1000, + system_prompt: "sp", + tools: [], + thinking_level: "default", + agent_state: "/tmp/a", + workspace: "/tmp/w", + }; +} + +describe("trace-service", () => { + let root: string; + let service: TraceService; + + beforeEach(async () => { + root = await makeTempRoot(); + service = new TraceService(root); + }); + afterEach(async () => { + await fs.rm(root, { recursive: true, force: true }); + }); + + it("messages:全部 index 文件按序拼接(跨日期目录)", async () => { + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + sessionMeta(metaPayload()), + userText("第一个文件"), + ]); + await writeTraceFile(root, P, A, "2026-07-06", S, 2, [ + sessionMeta(metaPayload()), + userText("第二个文件"), + ]); + const messages = await service.readMessages(P, A, S); + expect(messages).toHaveLength(4); + expect((messages[1]!.payload as { text: string }).text).toBe("第一个文件"); + expect((messages[3]!.payload as { text: string }).text).toBe("第二个文件"); + }); + + it("messages:容忍残缺末行", async () => { + const file = await writeTraceFile(root, P, A, "2026-07-05", S, 1, [userText("ok")]); + await fs.appendFile(file, '{"timestamp":"2026', "utf8"); + const messages = await service.readMessages(P, A, S); + expect(messages).toHaveLength(1); + }); + + it("traces 列表:index / 日期 / 大小 / mtime", async () => { + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [userText("a")]); + await writeTraceFile(root, P, A, "2026-07-06", S, 2, [userText("bb")]); + const files = await service.listTraceFiles(P, A, S); + expect(files.map((f) => f.index)).toEqual([1, 2]); + expect(files[0]!.date).toBe("2026-07-05"); + expect(files[0]!.sizeBytes).toBeGreaterThan(0); + expect(Date.parse(files[0]!.mtime)).not.toBeNaN(); + }); + + it("按行分页读取:offset/limit 与 total", async () => { + const messages = Array.from({ length: 10 }, (_, i) => userText(`m${i}`)); + await writeTraceFile(root, P, A, "2026-07-05", S, 1, messages); + const page = await service.readEvents(P, A, S, 1, 3, 4); + expect(page.total).toBe(10); + expect(page.offset).toBe(3); + expect(page.events).toHaveLength(4); + expect((page.events[0]!.payload as { text: string }).text).toBe("m3"); + const notFound = await service.readEvents(P, A, S, 99, 0, 10).catch((e: unknown) => e); + expect((notFound as { status: number }).status).toBe(404); + }); + + it("性能分析:Request 配对、工具耗时、reconnect / 压缩计数、Token 趋势", async () => { + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + sessionMeta(metaPayload()), + at("2026-07-05T10:00:00.000Z", userText("hi")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at("2026-07-05T10:00:03.000Z", requestEnd("timeout")), // reconnect +1 + at("2026-07-05T10:00:03.500Z", requestBegin()), + at( + "2026-07-05T10:00:04.000Z", + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc-1" }), + ), + at("2026-07-05T10:00:05.000Z", requestEnd("completed")), + at("2026-07-05T10:00:06.500Z", toolCallOutput({ output: "done", toolCallId: "tc-1" })), + at("2026-07-05T10:00:07.000Z", tokenUsage(counts(1000), counts(400))), + at( + "2026-07-05T10:00:08.000Z", + compactionBegin({ reason: "manual", mode: "summarize", context: 400, turns: 2 }), + ), + at( + "2026-07-05T10:00:09.000Z", + compactionEnd({ reason: "manual", mode: "summarize", status: "aborted" }), + ), + at("2026-07-05T10:00:10.000Z", abortEvent()), + at("2026-07-05T10:00:11.000Z", requestBegin()), // unclosed (process exited) + ]); + const analysis = await service.analyze(P, A, S, 1); + + expect(analysis.requests).toHaveLength(3); + expect(analysis.requests[0]!.status).toBe("timeout"); + expect(analysis.requests[0]!.durationMs).toBe(2000); + expect(analysis.requests[1]!.status).toBe("completed"); + expect(analysis.requests[1]!.durationMs).toBe(1500); + expect(analysis.requests[2]!.endTs).toBeUndefined(); + // A timeout is auto-reconnected by core within the same run, so the resent + // Request still belongs to **the same user turn**: req0(timeout) and + // req1(retry succeeded) are both Task 0 — they must not be split into two + // turns. Compaction interrupts continuation, so req2 starts a new turn. + expect(analysis.requests.map((r) => r.taskIndex)).toEqual([0, 0, 1]); + + expect(analysis.toolCalls).toHaveLength(1); + expect(analysis.toolCalls[0]!.name).toBe("exec_command"); + expect(analysis.toolCalls[0]!.durationMs).toBe(2500); + expect(analysis.toolCalls[0]!.stopReason).toBe("completed"); + + expect(analysis.reconnectCount).toBe(1); + expect(analysis.compactionCount).toBe(1); + expect(analysis.usageTrend).toEqual([ + { ts: "2026-07-05T10:00:07.000Z", requestTotal: 400, sessionTotal: 1000 }, + ]); + }); + + it("Task 上下文快照取该轮最后一次 Request,不是各次 Request 的累加", async () => { + // Two Requests within one Task (a tool call triggers another round): each + // input **re-carries the entire history**, so 60k → 65k is the context + // growing, not a 60k + 65k = 125k sum of usage. Summing them would double-count + // the context — a few rounds of tool calls would blow the context window past + // 100% and overflow the ring. + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + sessionMeta(metaPayload()), + at("2026-07-05T10:00:00.000Z", userText("hi")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at( + "2026-07-05T10:00:02.000Z", + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc-1" }), + ), + at("2026-07-05T10:00:03.000Z", requestEnd("completed")), + at("2026-07-05T10:00:03.100Z", tokenUsage(counts(60_000), buckets(50_000, 8_000, 2_000))), + at("2026-07-05T10:00:03.500Z", toolCallOutput({ output: "ok", toolCallId: "tc-1" })), + // Continuation round (same Task): context grows to 65k + at("2026-07-05T10:00:04.000Z", requestBegin()), + at("2026-07-05T10:00:06.000Z", requestEnd("completed")), + at("2026-07-05T10:00:06.100Z", tokenUsage(counts(65_000), buckets(58_000, 4_000, 3_000))), + ]); + const a = await service.analyze(P, A, S, 1); + + expect(a.requests.map((r) => r.taskIndex)).toEqual([0, 0]); // tool call → continues the same Task + expect(a.tasks).toHaveLength(1); + const t = a.tasks[0]!; + // Snapshot = the last Request (58k/4k/3k = 65k), not the sum of both (108k/12k/5k = 125k). + expect(t.context).toEqual({ cacheRead: 58_000, cacheWrite: 4_000, output: 3_000 }); + // The running total (used for Token/cost, with output doubling as the TPS + // numerator) IS summed: the three-bucket sum across both Requests. It and the + // snapshot above are two different measures — the frontend used to feed the + // running total into the ring as if it were the snapshot, which is how "two + // rounds of 60k/65k" ended up displaying as 125k. + expect(t.tokens).toEqual({ cacheRead: 108_000, cacheWrite: 12_000, output: 5_000 }); + expect(t.llmMs).toBe(2000 + 2000); + }); + + it("人工审批等待不计入 LLM 生成时长(TPS 分母)", async () => { + // core does `await approve(tc)` inside the streaming loop: until approval + // returns, the next chunk isn't consumed and request_end can't fire, so the + // entire human wait sits between request_begin and request_end. Without + // subtracting it, "2s of generation + 30s of approval wait" would drop the + // TPS to a fifteenth of the real value. + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + sessionMeta(metaPayload()), + at("2026-07-05T10:00:00.000Z", userText("hi")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at( + "2026-07-05T10:00:02.000Z", + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "tc-1" }), + ), + at("2026-07-05T10:00:32.000Z", approvalDecision("allow", "tc-1")), // human left it hanging for 30s + at("2026-07-05T10:00:33.000Z", requestEnd("completed")), + at("2026-07-05T10:00:33.100Z", tokenUsage(counts(1000), buckets(0, 0, 1_000))), + ]); + const a = await service.analyze(P, A, S, 1); + + const rq = a.requests[0]!; + expect(rq.durationMs).toBe(32_000); // wall clock: includes the approval wait + expect(rq.approvalWaitMs).toBe(30_000); + expect(rq.activeMs).toBe(2_000); // generation: 1s→2s (emits tool_call) + 32s→33s (wrap-up) + expect(a.tasks[0]!.llmMs).toBe(2_000); // the denominator uses only activeMs + expect(a.tasks[0]!.tokens.output).toBe(1_000); // → 500 tok/s, not 31 tok/s + }); + + it("压缩自成一轮:它的 TPS 归自己,不污染用户轮次;上下文快照仍只取非压缩 Request", async () => { + // Compaction's request_begin/end and token_usage all sit between + // compaction_begin and compaction_end (see core's context-engine summarize + // flow). Both sides of the Chat page exclude compaction, so Trace must use + // the same accounting, or the two pages would compute different TPS for the + // same Session. + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + sessionMeta(metaPayload()), + at("2026-07-05T10:00:00.000Z", userText("hi")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at("2026-07-05T10:00:03.000Z", requestEnd("completed")), + at("2026-07-05T10:00:03.100Z", tokenUsage(counts(30_000), buckets(20_000, 9_000, 1_000))), + at( + "2026-07-05T10:00:04.000Z", + compactionBegin({ reason: "context", mode: "summarize", context: 30_000, turns: 2 }), + ), + at("2026-07-05T10:00:05.000Z", requestBegin()), // the compaction request + at("2026-07-05T10:00:15.000Z", requestEnd("completed")), // slow: 10s + at("2026-07-05T10:00:15.100Z", tokenUsage(counts(32_000), buckets(29_000, 0, 3_000))), + at( + "2026-07-05T10:00:16.000Z", + compactionEnd({ reason: "context", mode: "summarize", status: "completed" }), + ), + ]); + const a = await service.analyze(P, A, S, 1); + + expect(a.requests[1]!.compaction).toBe(true); + expect(a.requests[0]!.compaction).toBeUndefined(); + expect(a.tasks.map((t) => t.taskIndex)).toEqual([0, 1]); + + // User turn: TPS counts only its own Request — none of the compaction's 3k output or 10s of generation time bleeds in. + const t = a.tasks[0]!; + expect(t.context).toEqual({ cacheRead: 20_000, cacheWrite: 9_000, output: 1_000 }); + expect(t.tokens.output).toBe(1_000); + expect(t.llmMs).toBe(2_000); + + // The compaction turn: it IS **a turn**, with its own TPS (how fast the summary was generated) — it shouldn't be blanked out as "—". + const ct = a.tasks[1]!; + expect(ct.llmMs).toBe(10_000); + // But it has no context snapshot: the tokens compaction consumes aren't the + // post-compaction context size (the frontend uses this to skip drawing a ring for the compaction turn). + expect(ct.context).toBeUndefined(); + // The running total is still recorded (output also serves as the compaction + // turn's own TPS numerator): compaction's tokens are genuinely paid for, so the cost must not be dropped. + expect(ct.tokens).toEqual({ cacheRead: 29_000, cacheWrite: 0, output: 3_000 }); + + // The compaction turn is flagged (the UI shows a "compaction" badge); its + // duration is measured from its request_begin (10:00:05) to compaction_end (10:00:16). + expect(ct.compaction).toBe(true); + expect(t.compaction).toBeUndefined(); + expect(Date.parse(ct.endTs) - Date.parse(ct.startTs)).toBe(11_000); + // Overall elapsed time = **the sum of every turn (including compaction turns)**, + // matching the same scope as the per-turn display — adding up the durations + // shown on each turn's card must equal the total. User turn 2.1s (request_begin + // 10:00:01 → token_usage 10:00:03.1) + compaction turn 11s. + expect(Date.parse(t.endTs) - Date.parse(t.startTs)).toBe(2_100); + expect(a.elapsedMs).toBe(2_100 + 11_000); + }); + + it("压缩请求重试耗尽(以 timeout 收尾)不把下一个用户轮次并进压缩 Task", async () => { + // The intersection of "timeout → continuation" and "compaction is its own + // turn": when a compaction request exhausts its retries and ends in timeout, + // it would be classified as continuing; compaction_end must clear that + // continuation flag, or the user turn after compaction would be folded into + // the compaction Task. + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + sessionMeta(metaPayload()), + at("2026-07-05T10:00:00.000Z", userText("hi")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at("2026-07-05T10:00:02.000Z", requestEnd("completed")), + at("2026-07-05T10:00:02.100Z", tokenUsage(counts(1000), buckets(0, 0, 500))), + at( + "2026-07-05T10:00:03.000Z", + compactionBegin({ reason: "context", mode: "summarize", context: 1000, turns: 1 }), + ), + at("2026-07-05T10:00:04.000Z", requestBegin()), // the compaction request + at("2026-07-05T10:00:05.000Z", requestEnd("timeout")), // retries exhausted, ends in timeout + at( + "2026-07-05T10:00:06.000Z", + compactionEnd({ reason: "context", mode: "summarize", status: "aborted" }), + ), + at("2026-07-05T10:00:07.000Z", userText("下一轮")), + at("2026-07-05T10:00:08.000Z", requestBegin()), + at("2026-07-05T10:00:10.000Z", requestEnd("completed")), + at("2026-07-05T10:00:10.100Z", tokenUsage(counts(1200), buckets(0, 0, 700))), + ]); + const a = await service.analyze(P, A, S, 1); + + // Task 0 = the first turn; Task 1 = compaction (its own turn); Task 2 = the + // user turn after compaction, which must not be folded into Task 1. + expect(a.requests.map((r) => r.taskIndex)).toEqual([0, 1, 2]); + // The compaction turn is also in the list (it has a start/end time and token + // cost, just no TPS or context snapshot) — if the table were built only from + // turns with "model output or a tool call", this kind of turn would disappear + // entirely and its events would get folded into the previous turn. + expect(a.tasks.map((t) => t.taskIndex)).toEqual([0, 1, 2]); + expect(a.tasks[1]!.tokens.output).toBe(0); // the compaction request timed out, producing nothing + expect(a.tasks[1]!.context).toBeUndefined(); + expect(a.tasks[2]!.tokens.output).toBe(700); + }); + + it("用户 Prompt 归入本轮消息区间,但用时从首个 request_begin 起算;空轮次照样在列表里", async () => { + // Message **attribution** (messageFrom/To) and **duration** (startTs/endTs) + // are two different things: the Prompt belongs to this turn's message range + // (the frontend uses this to list it on this turn's card), but the duration + // only looks at the LLM request — the start point is the first request_begin, + // and the user text's timestamp doesn't participate (the compaction summary + // `` is created during compaction but only persisted on the + // next run; using it as the start point would stretch the first turn out for + // no reason). Also, if the turn list were built only from turns with "a model + // segment or a tool span", a turn that fails outright with no output at all + // would disappear entirely, and its events would get folded into the previous + // turn — this must be backstopped by the server-side tasks logic too. + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + at("2026-07-05T10:00:00.000Z", sessionMeta(metaPayload())), + at("2026-07-05T10:00:00.000Z", userText("第一问")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at("2026-07-05T10:00:02.000Z", requestEnd("completed")), + at("2026-07-05T10:00:02.100Z", tokenUsage(counts(1000), buckets(0, 900, 100))), + // Second turn: the Prompt precedes the Request; this turn's request fails outright, with no model output or tool call at all. + at("2026-07-05T10:01:00.000Z", userText("第二问")), + at("2026-07-05T10:01:01.000Z", requestBegin()), + at("2026-07-05T10:01:04.000Z", requestEnd("failed")), + ]); + const a = await service.analyze(P, A, S, 1); + + expect(a.tasks.map((t) => t.taskIndex)).toEqual([0, 1]); // the empty turn is present too + // Message attribution: starting from "第二问" (index 5), it belongs to the second turn, not the tail of the previous one. + expect(a.tasks[0]!.messageTo).toBe(4); + expect(a.tasks[1]!.messageFrom).toBe(5); + // Duration start = request_begin (10:01:01), not the Prompt (10:01:00). + expect(a.tasks[1]!.startTs).toBe("2026-07-05T10:01:01.000Z"); + expect(a.tasks[1]!.endTs).toBe("2026-07-05T10:01:04.000Z"); + expect(a.tasks[1]!.tokens.output).toBe(0); // nothing was produced + expect(a.tasks[0]!.startTs).toBe("2026-07-05T10:00:01.000Z"); + + // Overall elapsed time = **the sum of each turn's duration**, not "last minus + // first": this example spans 64s overall, but 58s of that is the gap between + // turns where the user was thinking/away — not time the Agent spent working. + // Turn 0 = 1.1s, turn 1 = 3s, total 4.1s. + expect(Date.parse(a.tasks[0]!.endTs) - Date.parse(a.tasks[0]!.startTs)).toBe(1_100); + expect(Date.parse(a.tasks[1]!.endTs) - Date.parse(a.tasks[1]!.startTs)).toBe(3_000); + expect(a.elapsedMs).toBe(4_100); + }); + + it("上一轮以 timeout 收尾(重试耗尽)后,新的用户消息另起一轮", async () => { + // "timeout → continuation" holds only for **automatic retries within the + // same run**. Once retries are exhausted and the engine gives up, a message + // the user sends afterward starts a new turn — if continuation were still + // stuck at true, this turn would get folded into the failed one, mixing + // together both turns' messages, Tokens, TPS, and duration. A user Prompt + // always breaks continuation, regardless of how the previous turn wrapped up. + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + at("2026-07-05T10:00:00.000Z", sessionMeta(metaPayload())), + at("2026-07-05T10:00:00.000Z", userText("第一问")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at("2026-07-05T10:00:02.000Z", requestEnd("timeout")), // retries exhausted → gives up + at("2026-07-05T10:00:03.000Z", abortEvent()), + // A new user send + at("2026-07-05T10:01:00.000Z", userText("第二问")), + at("2026-07-05T10:01:01.000Z", requestBegin()), + at("2026-07-05T10:01:03.000Z", requestEnd("completed")), + at("2026-07-05T10:01:03.100Z", tokenUsage(counts(1000), buckets(0, 900, 100))), + ]); + const a = await service.analyze(P, A, S, 1); + expect(a.requests.map((r) => r.taskIndex)).toEqual([0, 1]); + expect(a.tasks.map((t) => t.taskIndex)).toEqual([0, 1]); + expect(a.tasks[1]!.startTs).toBe("2026-07-05T10:01:01.000Z"); // duration start = this turn's request_begin + expect(a.tasks[1]!.tokens.output).toBe(100); // the second turn's usage isn't folded into the first + }); + + it("一次发送含文本 + 多张图片:轮次归属从**第一条**消息起,不是最后一张图", async () => { + // One send = multiple messages (user text + some number of image_url). If the + // pending index were overwritten on every message, turn attribution would + // start from the last image, with the preceding text and images assigned to + // the previous turn — completely at odds with "the user clicked send once". + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + at("2026-07-05T10:00:00.000Z", sessionMeta(metaPayload())), + at("2026-07-05T10:00:00.000Z", userText("第一问")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at("2026-07-05T10:00:02.000Z", requestEnd("completed")), + at("2026-07-05T10:00:02.100Z", tokenUsage(counts(500), buckets(0, 400, 100))), + // Second send: text + two images + at("2026-07-05T10:01:00.000Z", userText("看这两张图")), + at("2026-07-05T10:01:00.500Z", imageUrlMessage("data:image/png;base64,AAAA")), + at("2026-07-05T10:01:01.000Z", imageUrlMessage("data:image/png;base64,BBBB")), + at("2026-07-05T10:01:02.000Z", requestBegin()), + at("2026-07-05T10:01:04.000Z", requestEnd("completed")), + at("2026-07-05T10:01:04.100Z", tokenUsage(counts(900), buckets(0, 800, 100))), + ]); + const a = await service.analyze(P, A, S, 1); + expect(a.tasks.map((t) => t.taskIndex)).toEqual([0, 1]); + // The message-attribution start = the text (index 5), not the last image (index 7) — all three messages belong to the second turn. + expect(a.tasks[1]!.messageFrom).toBe(5); + expect(a.tasks[0]!.messageTo).toBe(4); + expect(a.tasks[1]!.startTs).toBe("2026-07-05T10:01:02.000Z"); // duration start = request_begin + expect(a.tasks[0]!.endTs).toBe("2026-07-05T10:00:02.100Z"); // the second send's messages don't land in the first turn + }); + + it("消息逐条归属,不按时间戳猜:同一毫秒的「本轮回复 / 压缩开始 / 压缩 Prompt / 下轮 request」各归各轮", async () => { + // Automatic compaction triggered at turn wrap-up crams these messages into + // **the same millisecond**: this turn's last reply, compaction_begin, the + // compaction Prompt, and the compaction turn's request_begin. Attributing by + // time boundary simply can't separate them — this turn's reply would get + // assigned to the compaction turn. A single sequential server-side scan + // already knows which turn each message belongs to, so the frontend can just + // use messageFrom/messageTo for attribution. + const T = "2026-07-05T10:00:05.000Z"; // same millisecond + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + at("2026-07-05T10:00:00.000Z", sessionMeta(metaPayload())), + at("2026-07-05T10:00:00.000Z", userText("问")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at(T, assistantText("本轮的回复")), // ← this turn's own reply, same millisecond as the entries below + at(T, tokenUsage(counts(500), buckets(0, 400, 100))), + at(T, requestEnd("completed")), + at(T, compactionBegin({ reason: "context", mode: "summarize", context: 500, turns: 1 })), + at(T, userText("You have a partial transcript…")), // the compaction Prompt (also user text) + at(T, requestBegin()), // the compaction request + at("2026-07-05T10:00:25.000Z", requestEnd("completed")), // slow compaction: 20s + at( + "2026-07-05T10:00:25.000Z", + compactionEnd({ reason: "context", mode: "summarize", status: "completed" }), + ), + ]); + const a = await service.analyze(P, A, S, 1); + + expect(a.tasks.map((t) => t.taskIndex)).toEqual([0, 1]); + // Turn 0 = session_meta..request_end (index 0..5): **this turn's reply stays in this turn**. + expect([a.tasks[0]!.messageFrom, a.tasks[0]!.messageTo]).toEqual([0, 5]); + // Turn 1 = compaction_begin..compaction_end (index 6..10): the compaction + // Prompt belongs to the compaction turn, not the tail of the previous one. + expect([a.tasks[1]!.messageFrom, a.tasks[1]!.messageTo]).toEqual([6, 10]); + + // Duration is attributed to each turn separately: this turn is 4s + // (request_begin 10:00:01 → request_end 10:00:05, excluding compaction's 20s), + // the compaction turn is 20s (the compaction request's request_begin → + // compaction_end, both starting at 10:00:05). + expect(Date.parse(a.tasks[0]!.endTs) - Date.parse(a.tasks[0]!.startTs)).toBe(4_000); + expect(Date.parse(a.tasks[1]!.endTs) - Date.parse(a.tasks[1]!.startTs)).toBe(20_000); + }); + + it("Agent 级逐级浏览:日期倒序、Session 倒序、文件 index 升序", async () => { + const s2 = "session-2026-07-06-09-00-00-11112222"; + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [userText("a")]); + await writeTraceFile(root, P, A, "2026-07-05", S, 2, [userText("b")]); + await writeTraceFile(root, P, A, "2026-07-06", s2, 1, [userText("c")]); + const res = await service.agentTraces(P, A); + expect(res.dates.map((d) => d.date)).toEqual(["2026-07-06", "2026-07-05"]); + expect(res.dates[1]!.sessions[0]!.sessionId).toBe(S); + expect(res.dates[1]!.sessions[0]!.files.map((f) => f.index)).toEqual([1, 2]); + }); + + it("无 Trace 时各接口返回空", async () => { + expect(await service.readMessages(P, A, S)).toEqual([]); + expect(await service.listTraceFiles(P, A, S)).toEqual([]); + expect((await service.agentTraces(P, A)).dates).toEqual([]); + }); + it("执行时间线:模型串行分段(起点=上一事件)、工具审批/执行两阶段、下一轮以 request_begin 起算", async () => { + const T = (sec: string) => `2026-07-05T10:00:${sec}Z`; + await writeTraceFile(root, P, A, "2026-07-05", S, 7, [ + sessionMeta(metaPayload()), + at(T("00.000"), userText("q")), // the user input is sent instantly, so it occupies no segment + at(T("01.000"), requestBegin()), + at(T("03.000"), thinkingMessage("想", "completed")), + at(T("04.000"), toolCall({ name: "exec_command", arguments: "{}", toolCallId: "t1" })), + // Two async tools: t1 is already in approval/execution while the model keeps decoding t2 + at(T("04.500"), toolCall({ name: "read_file", arguments: "{}", toolCallId: "t2" })), + at(T("05.000"), requestEnd("completed")), + at(T("05.500"), approvalDecision("allow", "t1")), + at(T("06.000"), approvalDecision("allow", "t2")), + at(T("07.000"), toolCallOutput({ output: "o1", toolCallId: "t1" })), + at(T("08.000"), toolCallOutput({ output: "o2", toolCallId: "t2" })), + // The model starts the next round only after all outputs are back: the new segment is anchored on request_begin + at(T("08.500"), requestBegin()), + at(T("10.000"), assistantText("答")), + at(T("10.100"), requestEnd("completed")), + ]); + const a = await service.analyze(P, A, S, 7); + + // A single user turn containing two rounds of Requests (the first calls a + // tool, the second produces the answer) is merged into the same Task (taskIndex 0). + expect(a.modelSegments).toEqual([ + { kind: "thinking", startTs: T("01.000"), endTs: T("03.000"), taskIndex: 0 }, + { + kind: "tool_call", + startTs: T("03.000"), + endTs: T("04.000"), + toolCallId: "t1", + name: "exec_command", + taskIndex: 0, + }, + { + kind: "tool_call", + startTs: T("04.000"), + endTs: T("04.500"), + toolCallId: "t2", + name: "read_file", + taskIndex: 0, + }, + { kind: "text", startTs: T("08.500"), endTs: T("10.000"), taskIndex: 0 }, + ]); + expect(a.toolSpans).toEqual([ + { + toolCallId: "t1", + name: "exec_command", + callTs: T("04.000"), + approvalTs: T("05.500"), + decision: "allow", + outputTs: T("07.000"), + stopReason: "completed", + taskIndex: 0, + }, + { + toolCallId: "t2", + name: "read_file", + callTs: T("04.500"), + approvalTs: T("06.000"), + decision: "allow", + outputTs: T("08.000"), + stopReason: "completed", + taskIndex: 0, + }, + ]); + }); + + it("Task 分组:模型出纯文本后不再调工具 → 下一用户轮次进入新 Task(taskIndex 递增)", async () => { + const T = (sec: string) => `2026-07-05T10:01:${sec}Z`; + await writeTraceFile(root, P, A, "2026-07-05", S, 9, [ + sessionMeta(metaPayload()), + // Task 0: one round calls a tool + one round produces the answer. + at(T("00.000"), userText("q1")), + at(T("01.000"), requestBegin()), + at(T("02.000"), toolCall({ name: "read_file", arguments: "{}", toolCallId: "t1" })), + at(T("02.500"), requestEnd("completed")), + at(T("03.000"), toolCallOutput({ output: "o", toolCallId: "t1" })), + at(T("03.500"), requestBegin()), + at(T("04.000"), assistantText("答1")), + at(T("04.500"), requestEnd("completed")), + // Task 1: a new user turn (the previous turn ended in plain text, not a continuation). + at(T("20.000"), userText("q2")), + at(T("21.000"), requestBegin()), + at(T("22.000"), assistantText("答2")), + at(T("22.500"), requestEnd("completed")), + ]); + const a = await service.analyze(P, A, S, 9); + expect(a.modelSegments.map((s) => s.taskIndex)).toEqual([0, 0, 1]); + expect(a.toolSpans.map((s) => s.taskIndex)).toEqual([0]); + // The first round calls a tool → the continuation round stays in Task 0; after ending in plain text, the new user turn goes into Task 1. + expect(a.requests.map((r) => r.taskIndex)).toEqual([0, 0, 1]); + }); + + // Compaction is its own turn: the previous turn called a tool and would + // otherwise "continue", but compaction_begin breaks that continuation, so the + // compaction request lands on a new taskIndex. A successful compaction splits + // the Trace into a new file, so this file ends at compaction_end. + it("Task 分组:压缩请求不并入上一轮(成功压缩,本文件在 compaction_end 收尾)", async () => { + const T = (sec: string) => `2026-07-05T10:02:${sec}Z`; + await writeTraceFile(root, P, A, "2026-07-05", S, 10, [ + sessionMeta(metaPayload()), + // Task 0: calls a tool, which would normally continue the turn. + at(T("00.000"), userText("q")), + at(T("01.000"), requestBegin()), + at(T("02.000"), toolCall({ name: "read_file", arguments: "{}", toolCallId: "t1" })), + at(T("02.500"), requestEnd("completed")), + at(T("03.000"), toolCallOutput({ output: "o", toolCallId: "t1" })), + // Task 1: the compaction request. + at( + T("04.000"), + compactionBegin({ reason: "context", mode: "summarize", context: 1, turns: 1 }), + ), + at(T("04.500"), requestBegin()), + at(T("05.000"), assistantText("…")), + at(T("05.500"), requestEnd("completed")), + at(T("06.000"), compactionEnd({ reason: "context", mode: "summarize", status: "completed" })), + ]); + const a = await service.analyze(P, A, S, 10); + expect(a.modelSegments.map((s) => s.taskIndex)).toEqual([0, 1]); + expect(a.toolSpans.map((s) => s.taskIndex)).toEqual([0]); + expect(a.requests.map((r) => r.taskIndex)).toEqual([0, 1]); + expect(a.compactionCount).toBe(1); + }); + + // A failed compaction doesn't split the file: the continuation request is + // still in the same file, and it starts a new turn (the compaction request + // doesn't call a tool, and request_end has already broken continuation). + it("Task 分组:压缩失败后的续跑请求另起一轮", async () => { + const T = (sec: string) => `2026-07-05T10:03:${sec}Z`; + await writeTraceFile(root, P, A, "2026-07-05", S, 11, [ + sessionMeta(metaPayload()), + at(T("00.000"), userText("q")), + at(T("01.000"), requestBegin()), + at(T("02.000"), toolCall({ name: "read_file", arguments: "{}", toolCallId: "t1" })), + at(T("02.500"), requestEnd("completed")), + at(T("03.000"), toolCallOutput({ output: "o", toolCallId: "t1" })), + at( + T("04.000"), + compactionBegin({ reason: "context", mode: "summarize", context: 1, turns: 1 }), + ), + at(T("04.500"), requestBegin()), + at(T("05.000"), assistantText("坏摘要")), + at(T("05.500"), requestEnd("failed")), + at(T("06.000"), compactionEnd({ reason: "context", mode: "summarize", status: "failed" })), + at(T("07.000"), requestBegin()), + at(T("08.000"), assistantText("答")), + at(T("08.500"), requestEnd("completed")), + ]); + const a = await service.analyze(P, A, S, 11); + expect(a.modelSegments.map((s) => s.taskIndex)).toEqual([0, 1, 2]); + expect(a.toolSpans.map((s) => s.taskIndex)).toEqual([0]); + }); + + it("中断补偿 tool_call(stop_reason 非 completed)不入时间线泳道(不生成幻影执行段)", async () => { + const T = (sec: string) => `2026-07-05T10:00:${sec}Z`; + await writeTraceFile(root, P, A, "2026-07-05", S, 8, [ + sessionMeta(metaPayload()), + at(T("00.000"), userText("q")), + at(T("01.000"), requestBegin()), + at( + T("02.000"), + toolCall({ + name: "exec_command", + arguments: "{}", + toolCallId: "t1", + stopReason: "timeout", + }), + ), + at(T("02.500"), requestEnd("timeout")), + at(T("03.000"), requestBegin()), + at(T("04.000"), toolCall({ name: "read_file", arguments: "{}", toolCallId: "t2" })), + at(T("04.500"), toolCallOutput({ output: "ok", toolCallId: "t2" })), + at(T("05.000"), requestEnd("completed")), + ]); + const a = await service.analyze(P, A, S, 8); + // A phantom call gets no lane: only t2 goes into toolSpans. + expect(a.toolSpans.map((sp) => sp.toolCallId)).toEqual(["t2"]); + // But the duration list still records t1 (flagged with the interrupted status). + expect(a.toolCalls.find((c) => c.toolCallId === "t1")?.stopReason).toBe("timeout"); + }); +}); diff --git a/packages/server/test/trace-subagent-expand.test.ts b/packages/server/test/trace-subagent-expand.test.ts new file mode 100644 index 0000000..487bb19 --- /dev/null +++ b/packages/server/test/trace-subagent-expand.test.ts @@ -0,0 +1,163 @@ +/** + * Sub-session expansion in historical messages (conversation history). + * + * The parent Trace records only a single `subagent` pointer event at the spawn + * point (holding just the child Session id); the content stays in the child + * Session's own Trace. `TraceService.readMessages` uses the pointer to locate + * the child Trace within the Project, recursively reads back the child messages, + * and attaches the origin chain — otherwise, after a page refresh the frontend + * would have no way to reattach the sub-session to its run_subagent tool card, + * and the sub-session view would disappear entirely. + */ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import fs from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { + assistantText, + sessionMeta, + subagentEvent, + toolCall, + toolCallOutput, + userText, +} from "@prismshadow/penguin-core"; +import type { OmniMessage, SessionMetaPayload } from "@prismshadow/penguin-core"; +import { TraceService } from "../src/services/trace-service.js"; + +const PROJECT = "proj"; +const PARENT_AGENT = "default_agent"; +const CHILD_AGENT = "worker"; +const PARENT = "session-2026-07-09-10-00-00-aaaa0001"; +const CHILD = "session-2026-07-09-10-00-01-bbbb0002"; +const GRANDCHILD = "session-2026-07-09-10-00-02-cccc0003"; + +function meta(sessionId: string, agentId: string): OmniMessage { + const payload: SessionMetaPayload = { + session_id: sessionId, + model_id: "m", + provider: "custom", + model_context_window: 1000, + system_prompt: "", + tools: [], + thinking_level: "medium", + agent_state: `/root/${PROJECT}/${agentId}/agent_state`, + workspace: "/tmp/w", + }; + return sessionMeta(payload); +} + +async function writeTrace( + root: string, + agentId: string, + sessionId: string, + messages: OmniMessage[], +): Promise { + const dir = path.join(root, PROJECT, "agents", agentId, "traces", "2026-07-09"); + await fs.mkdir(dir, { recursive: true }); + await fs.writeFile( + path.join(dir, `${sessionId}_001.jsonl`), + messages.map((m) => JSON.stringify(m)).join("\n") + "\n", + "utf8", + ); +} + +describe("TraceService.readMessages — 子会话展开", () => { + let root: string; + let svc: TraceService; + + beforeEach(async () => { + root = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-trace-expand-")); + svc = new TraceService(root); + }); + afterEach(async () => { + await fs.rm(root, { recursive: true, force: true }); + }); + + it("按指针就地插入子会话消息,并补上 origin 链(子 Agent 按 Session id 在 Project 内定位)", async () => { + await writeTrace(root, PARENT_AGENT, PARENT, [ + meta(PARENT, PARENT_AGENT), + userText("跑个子 agent"), + toolCall({ name: "run_subagent", arguments: "{}", toolCallId: "c1" }), + subagentEvent(CHILD), // pointer: holds only the child Session id + toolCallOutput({ output: "done", toolCallId: "c1" }), + ]); + await writeTrace(root, CHILD_AGENT, CHILD, [ + meta(CHILD, CHILD_AGENT), + assistantText("子会话回答"), + ]); + + const msgs = await svc.readMessages(PROJECT, PARENT_AGENT, PARENT); + const texts = msgs.map((m) => (m.payload as { text?: string; type?: string }).text ?? m.type); + expect(texts).toEqual([ + "session_meta", + "跑个子 agent", + "model_msg", // tool_call + "session_meta", // the child session's own meta (replaces the pointer, matching the live-stream forwarding shape) + "子会话回答", + "model_msg", // tool_call_output + ]); + + // The pointer event is replaced by the child Trace content and no longer appears; all child messages carry an origin chain. + expect(msgs.some((m) => (m.payload as { type?: string }).type === "subagent")).toBe(false); + const nested = msgs.filter((m) => m.origin !== undefined); + expect(nested).toHaveLength(2); + for (const m of nested) expect(m.origin).toEqual([CHILD]); + }); + + it("递归展开孙会话,origin 链逐层前缀", async () => { + await writeTrace(root, PARENT_AGENT, PARENT, [ + meta(PARENT, PARENT_AGENT), + subagentEvent(CHILD), + ]); + await writeTrace(root, CHILD_AGENT, CHILD, [ + meta(CHILD, CHILD_AGENT), + subagentEvent(GRANDCHILD), + ]); + await writeTrace(root, CHILD_AGENT, GRANDCHILD, [ + meta(GRANDCHILD, CHILD_AGENT), + assistantText("孙会话回答"), + ]); + + const msgs = await svc.readMessages(PROJECT, PARENT_AGENT, PARENT); + const deepest = msgs.find((m) => (m.payload as { text?: string }).text === "孙会话回答"); + expect(deepest?.origin).toEqual([CHILD, GRANDCHILD]); + expect(msgs.filter((m) => m.type === "session_meta")).toHaveLength(3); + }); + + it("循环指针(指向自身/祖先)不展开,保留指针事件", async () => { + // Never produced by normal operation (core only writes a direct child-session + // pointer at the spawn point); this guards against expansion runaway on a + // tampered/corrupted Trace. + await writeTrace(root, PARENT_AGENT, PARENT, [ + meta(PARENT, PARENT_AGENT), + subagentEvent(PARENT), // self-reference + subagentEvent(CHILD), + ]); + await writeTrace(root, CHILD_AGENT, CHILD, [ + meta(CHILD, CHILD_AGENT), + subagentEvent(PARENT), // points back to an ancestor + assistantText("子会话回答"), + ]); + + const msgs = await svc.readMessages(PROJECT, PARENT_AGENT, PARENT); + // Self-referencing and ancestor-pointing pointers are kept as-is; CHILD expands normally exactly once. + const pointers = msgs.filter((m) => (m.payload as { type?: string }).type === "subagent"); + expect(pointers).toHaveLength(2); + expect(msgs.filter((m) => m.type === "session_meta")).toHaveLength(2); + expect(msgs.filter((m) => (m.payload as { text?: string }).text === "子会话回答")).toHaveLength( + 1, + ); + }); + + it("子 Trace 缺失时保留指针事件(子会话内容无从恢复,但派生记录不凭空消失)", async () => { + await writeTrace(root, PARENT_AGENT, PARENT, [ + meta(PARENT, PARENT_AGENT), + subagentEvent(CHILD), + ]); + + const msgs = await svc.readMessages(PROJECT, PARENT_AGENT, PARENT); + expect(msgs).toHaveLength(2); + expect(msgs[1]!.type).toBe("event_msg"); + expect(msgs[1]!.payload).toMatchObject({ type: "subagent", session_id: CHILD }); + }); +}); diff --git a/packages/server/test/usage.test.ts b/packages/server/test/usage.test.ts new file mode 100644 index 0000000..1dafa77 --- /dev/null +++ b/packages/server/test/usage.test.ts @@ -0,0 +1,280 @@ +/** + * Unit tests for usage persistence and statistics: origin→model attribution, + * summary buckets / group aggregation / trend queries, cost computed **on the + * fly** (only Tokens are persisted; cost is priced against current pricing at + * query time, no pricing → NULL + hasUncosted; a later price update is + * reflected immediately), and the status → success-rate pipeline (aborted + * doesn't count as a model failure). + */ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { sessionMeta, tokenUsage, withOrigin } from "@prismshadow/penguin-core"; +import type { SessionMetaPayload, TokenCounts } from "@prismshadow/penguin-core"; +import { ORIGIN_MODELS_MAX, UsageRecorder } from "../src/runtime/usage-recorder.js"; +import { ErrorsRepo } from "../src/db/repos/errors.js"; +import { UsageRepo } from "../src/db/repos/usage.js"; +import { UsageService } from "../src/services/usage-service.js"; +import type { PricingRates } from "../src/services/usage-service.js"; +import { openDatabase } from "../src/db/database.js"; +import { formatLocalDate } from "../src/internal/dates.js"; +import type { DatabaseSync } from "node:sqlite"; + +const CTX = { + projectId: "project-x", + agentId: "agent-x", + sessionId: "session-main", + modelId: "main-model", + provider: "custom", +}; + +function counts(total: number): TokenCounts { + return { cache_read: 100, cache_write: 10, output: 5, total }; +} + +function meta(sessionId: string, modelId: string, provider = "custom"): SessionMetaPayload { + return { + session_id: sessionId, + provider, + model_id: modelId, + model_context_window: 100000, + system_prompt: "", + tools: [], + thinking_level: "default", + agent_state: "/tmp/x", + workspace: "/tmp/w", + }; +} + +describe("usage-recorder", () => { + let db: DatabaseSync; + let repo: UsageRepo; + + beforeEach(() => { + db = openDatabase(":memory:"); + repo = new UsageRepo(db); + }); + afterEach(() => db.close()); + + it("token_usage → 一行记录(request 桶,只落 Token 不落成本)", async () => { + const rec = new UsageRecorder(repo); + await rec.record(CTX, tokenUsage(counts(1000), counts(115))); + const rows = db.prepare("SELECT * FROM usage_records").all(); + expect(rows).toHaveLength(1); + const row = rows[0]!; + expect(row.session_id).toBe("session-main"); + expect(row.origin_session_id).toBeNull(); + expect(row.model_id).toBe("main-model"); + expect(row.total).toBe(115); // taken from request.total + expect(row.cache_read).toBe(100); + }); + + it("子会话 session_meta 登记 origin→model 映射;token_usage 按映射归因", async () => { + const rec = new UsageRecorder(repo); + const childMeta = withOrigin( + sessionMeta(meta("session-child", "child-model")), + "session-child", + ); + await rec.record(CTX, childMeta); + await rec.record(CTX, withOrigin(tokenUsage(counts(50), counts(50)), "session-child")); + const row = db.prepare("SELECT * FROM usage_records").get()!; + expect(row.session_id).toBe("session-main"); // attributed to its owning main Session + expect(row.origin_session_id).toBe("session-child"); + expect(row.model_id).toBe("child-model"); + }); + + it("origin 映射缺失时回退主 Session Model", async () => { + const rec = new UsageRecorder(repo); + await rec.record(CTX, withOrigin(tokenUsage(counts(5), counts(5)), "session-unknown")); + const row = db.prepare("SELECT model_id FROM usage_records").get()!; + expect(row.model_id).toBe("main-model"); + }); + + it("非 token_usage 消息为 no-op", async () => { + const rec = new UsageRecorder(repo); + await rec.record(CTX, sessionMeta(meta("session-main", "main-model"))); + expect(db.prepare("SELECT COUNT(*) AS n FROM usage_records").get()!.n).toBe(0); + }); + + it("origin 映射有上限:超限淘汰最早登记项,被淘汰的回退主 Session Model", async () => { + const rec = new UsageRecorder(repo); + for (let i = 0; i <= ORIGIN_MODELS_MAX; i++) { + // ORIGIN_MODELS_MAX + 1 entries total: the earliest, sub-0, gets evicted. + await rec.record(CTX, withOrigin(sessionMeta(meta(`sub-${i}`, "sub-model")), `sub-${i}`)); + } + await rec.record(CTX, withOrigin(tokenUsage(counts(5), counts(5)), "sub-0")); + await rec.record(CTX, withOrigin(tokenUsage(counts(5), counts(5)), `sub-${ORIGIN_MODELS_MAX}`)); + const rows = db.prepare("SELECT model_id FROM usage_records ORDER BY id").all(); + expect(rows[0]!.model_id).toBe("main-model"); // evicted → falls back + expect(rows[1]!.model_id).toBe("sub-model"); // still mapped + }); +}); + +describe("usage-service(成本实时折算)", () => { + let db: DatabaseSync; + let repo: UsageRepo; + let service: (now: Date) => UsageService; + /** Mutable pricing table: simulates a "price added later" — change the price after inserting a record, and the query reflects it immediately. */ + let pricing: Record; + + // The pricing lookup callback takes three params (projectId, provider, modelId): locates the price via the paired reference. + const lookup = async (_p: string, _provider: string, modelId: string) => pricing[modelId]; + + beforeEach(() => { + db = openDatabase(":memory:"); + repo = new UsageRepo(db); + const errors = new ErrorsRepo(db); + service = (now: Date) => new UsageService(repo, errors, lookup, () => now); + pricing = { m1: { cacheRead: 0.3, cacheWrite: 3.75, output: 15 } }; + }); + afterEach(() => db.close()); + + // Fixed Tokens per row: cacheRead=10, cacheWrite=1, output=5 → the per-row cost for m1 + const ROW_COST = (10 * 0.3 + 1 * 3.75 + 5 * 15) / 1e6; + + function insert(date: string, opts: Partial[0]> = {}): void { + repo.insert({ + ts: `${date}T00:00:00.000Z`, + date, + projectId: "p1", + agentId: "a1", + sessionId: "s1", + originSessionId: null, + modelId: "m1", + provider: "custom", + cacheRead: 10, + cacheWrite: 1, + output: 5, + total: 100, + ...opts, + }); + } + + it("汇总卡片:今日 / 近 7 天 / 累计;无 pricing 的 Model 标记 hasUncosted", async () => { + const now = new Date("2026-07-06T10:00:00"); + const today = formatLocalDate(now); + insert(today); + insert("2026-07-03"); // within the last 7 days + insert("2026-06-01", { modelId: "m-unpriced" }); // only in the cumulative total; this Model has no pricing + const svc = service(now); + const res = await svc.query("p1", { groupBy: "date" }); + expect(res.summary.today.total).toBe(100); + expect(res.summary.today.requests).toBe(1); + expect(res.summary.last7d.total).toBe(200); + expect(res.summary.total.total).toBe(300); + expect(res.summary.total.cost).toBeCloseTo(ROW_COST * 2, 12); + expect(res.summary.total.hasUncosted).toBe(true); + expect(res.summary.last7d.hasUncosted).toBe(false); + }); + + it("价格后补:插入时无 pricing,配置价格后再查询即计价", async () => { + const now = new Date("2026-07-06T10:00:00"); + insert("2026-07-06", { modelId: "m-late" }); + const svc = service(now); + + const before = await svc.query("p1", { groupBy: "date" }); + expect(before.summary.total.cost).toBeNull(); + expect(before.summary.total.hasUncosted).toBe(true); + + pricing["m-late"] = { cacheRead: 1, cacheWrite: 1, output: 1 }; + const after = await svc.query("p1", { groupBy: "date" }); + expect(after.summary.total.cost).toBeCloseTo((10 + 1 + 5) / 1e6, 12); + expect(after.summary.total.hasUncosted).toBe(false); + }); + + it("分组聚合:date 按日期倒序;agent/model/session 维度与 agentId 下钻;跨 Model 折叠", async () => { + const now = new Date("2026-07-06T10:00:00"); + pricing.m2 = { cacheRead: 1, cacheWrite: 1, output: 1 }; + insert("2026-07-05", { agentId: "a1", sessionId: "s1", modelId: "m1" }); + insert("2026-07-06", { agentId: "a2", sessionId: "s2", modelId: "m2", total: 300 }); + insert("2026-07-06", { agentId: "a2", sessionId: "s3", modelId: "m1" }); + const svc = service(now); + + const byDate = await svc.query("p1", { groupBy: "date" }); + expect(byDate.groups.map((g) => g.key)).toEqual(["2026-07-06", "2026-07-05"]); + expect(byDate.groups[0]!.total).toBe(400); + expect(byDate.groups[0]!.requests).toBe(2); + // Same date, folded across Models: one m2 row + one m1 row. + expect(byDate.groups[0]!.cost).toBeCloseTo((10 + 1 + 5) / 1e6 + ROW_COST, 12); + + const byAgent = await svc.query("p1", { groupBy: "agent" }); + expect(byAgent.groups[0]!.key).toBe("a2"); // sorted by Token count descending + + const bySession = await svc.query("p1", { groupBy: "session", agentId: "a2" }); + expect(bySession.groups.map((g) => g.key).sort()).toEqual(["s2", "s3"]); + + // Not visible from another Project. + const other = await svc.query("p-other", { groupBy: "date" }); + expect(other.groups).toEqual([]); + }); + + it("from/to 过滤分组;趋势固定近 30 天窗口", async () => { + const now = new Date("2026-07-06T10:00:00"); + insert("2026-07-06"); + insert("2026-06-20"); + insert("2026-05-01"); // outside the 30-day window + const svc = service(now); + const res = await svc.query("p1", { + groupBy: "date", + from: "2026-07-01", + to: "2026-07-31", + }); + expect(res.groups.map((g) => g.key)).toEqual(["2026-07-06"]); + expect(res.trend.map((p) => p.date)).toEqual(["2026-06-20", "2026-07-06"]); + expect(res.trend[1]!.cost).toBeCloseTo(ROW_COST, 12); + }); + + // —— status → success-rate pipeline —— + + it("成功率:completed / 非 aborted 请求,失败细分随行带出", async () => { + const now = new Date("2026-07-06T10:00:00"); + for (let i = 0; i < 7; i++) insert("2026-07-06"); + insert("2026-07-06", { status: "failed", total: 0 }); + insert("2026-07-06", { status: "timeout", total: 0 }); + insert("2026-07-06", { status: "malformed", total: 0 }); + const res = await service(now).query("p1", { groupBy: "date" }); + + const m1 = res.success.find((s) => s.modelId === "m1")!; + expect(m1).toMatchObject({ + completed: 7, + total: 10, + aborted: 0, + failed: 1, + timeout: 1, + malformed: 1, + }); + }); + + it("回归:aborted(用户点「停止」)不是模型失败——不进分母,成功率不因中断下降", async () => { + const now = new Date("2026-07-06T10:00:00"); + for (let i = 0; i < 8; i++) insert("2026-07-06"); + // The user clicked "Stop" twice: under the old accounting, the success rate would drop to 8/10 = 80%. + insert("2026-07-06", { status: "aborted", total: 0 }); + insert("2026-07-06", { status: "aborted", total: 0 }); + const res = await service(now).query("p1", { groupBy: "date" }); + + const m1 = res.success.find((s) => s.modelId === "m1")!; + expect(m1.completed).toBe(8); + expect(m1.total).toBe(8); // denominator excludes aborted + expect(m1.aborted).toBe(2); // but the info isn't lost + expect(m1.completed / m1.total).toBe(1); // 100%, no longer dragged down by aborts + + // A real failure still counts: add one more failed → 8/9. + insert("2026-07-06", { status: "failed", total: 0 }); + const after = await service(now).query("p1", { groupBy: "date" }); + expect(after.success.find((s) => s.modelId === "m1")!.total).toBe(9); + }); + + it("成功率不受 model 过滤、仍受 agent 与日期过滤(图展示全部 Model)", async () => { + const now = new Date("2026-07-06T10:00:00"); + insert("2026-07-06", { modelId: "m1", agentId: "a1" }); + insert("2026-07-06", { modelId: "m2", agentId: "a2", status: "failed", total: 0 }); + const svc = service(now); + + // Filtering by m1: the success-rate chart still lists m2 (for comparison), unaffected by the filter. + const filtered = await svc.query("p1", { groupBy: "date", modelId: "m1" }); + expect(filtered.success.map((s) => s.modelId).sort()).toEqual(["m1", "m2"]); + + // Filtering by a1: m2's requests belong to a2 and are excluded. + const byAgent = await svc.query("p1", { groupBy: "date", agentId: "a1" }); + expect(byAgent.success.map((s) => s.modelId)).toEqual(["m1"]); + }); +}); diff --git a/packages/server/test/vault.test.ts b/packages/server/test/vault.test.ts new file mode 100644 index 0000000..5f0bfc6 --- /dev/null +++ b/packages/server/test/vault.test.ts @@ -0,0 +1,162 @@ +/** + * Integration tests for the Vault environment variable routes (Agent-level + * agent_state/.vault.toml): GET masks values (plaintext + * is never sent), PUT is owner-only, whole-table replace semantics (omitting + * value keeps the original, an absent key is deleted, a new key must supply a + * value), 400 on key/shape validation, 404 for a nonexistent Agent, and vaults + * of different Agents are independent of each other. + */ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import type { ProjectCreateResponse, VaultResponse } from "../src/api/types.js"; +import { apiClient, createTestApp, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +describe("vault api", () => { + let t: TestApp; + let owner: ReturnType; + let member: ReturnType; + let outsider: ReturnType; + let projectId: string; + let vaultPath: string; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner_a"); + const b = await provisionUser(t.app, "member_b"); + const c = await provisionUser(t.app, "outsider_c"); + owner = apiClient(t.app, a.cookie); + member = apiClient(t.app, b.cookie); + outsider = apiClient(t.app, c.cookie); + const created = (await ( + await owner.post("/api/projects", { projectId: "owner_a-vault", name: "vault 项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + vaultPath = `/api/projects/${projectId}/agents/default_agent/vault`; + const add = await owner.post(`/api/projects/${projectId}/members`, { userId: "member_b" }); + expect(add.status).toBe(201); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("GET 掩码且明文不下发;member 可读、外人 404;空 vault 返回空表", async () => { + // Not configured yet: an empty table. + const empty = (await (await owner.get(vaultPath)).json()) as VaultResponse; + expect(empty.entries).toEqual([]); + + const put = await owner.put(vaultPath, { + entries: [ + { key: "OPENAI_API_KEY", value: "sk-vault-secret-123456" }, + { key: "SHORT", value: "sk-11-chars" }, + ], + }); + expect(put.status).toBe(200); + // The PUT response likewise contains only masked values. + expect(JSON.stringify(await put.json())).not.toContain("sk-vault-secret-123456"); + + const res = await member.get(vaultPath); + expect(res.status).toBe(200); + const body = (await res.json()) as VaultResponse; + expect(body.entries).toEqual([ + { key: "OPENAI_API_KEY", valueMasked: "sk-v…3456" }, + { key: "SHORT", valueMasked: "***" }, // ≤12 chars are masked entirely (first4…last4 would expose over half) + ]); + expect(JSON.stringify(body)).not.toContain("sk-vault-secret-123456"); + expect(JSON.stringify(body)).not.toContain("sk-11-chars"); + + expect((await outsider.get(vaultPath)).status).toBe(404); + }); + + it("PUT 仅 owner:member 403、外人 404", async () => { + expect((await member.put(vaultPath, { entries: [{ key: "K", value: "v" }] })).status).toBe(403); + expect((await outsider.put(vaultPath, { entries: [{ key: "K", value: "v" }] })).status).toBe( + 404, + ); + }); + + it("整表替换:value 省略保留原值、缺席的键删除、新键缺 value 400", async () => { + await owner.put(vaultPath, { + entries: [ + { key: "KEEP_ME", value: "keep-secret-000111" }, + { key: "DROP_ME", value: "drop-secret" }, + ], + }); + + // KEEP_ME only sends back the key name (keeping the original value); DROP_ME is absent (deleted). + const second = (await ( + await owner.put(vaultPath, { entries: [{ key: "KEEP_ME" }] }) + ).json()) as VaultResponse; + expect(second.entries).toEqual([{ key: "KEEP_ME", valueMasked: "keep…0111" }]); + + // A new key without value → 400 (there's no original value to keep). + const bad = await owner.put(vaultPath, { + entries: [{ key: "KEEP_ME" }, { key: "BRAND_NEW" }], + }); + expect(bad.status).toBe(400); + + // Clear the whole table. + const cleared = (await (await owner.put(vaultPath, { entries: [] })).json()) as VaultResponse; + expect(cleared.entries).toEqual([]); + }); + + it("vault 更新不动 models/credential 等其他配置", async () => { + await owner.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "custom", modelId: "m-1" }, + models: [{ provider: "custom", modelId: "m-1", apiKey: "sk-model-key-999888" }], + }); + await owner.put(vaultPath, { entries: [{ key: "K1", value: "v1-secret" }] }); + + const models = await owner.get(`/api/projects/${projectId}/models`); + const modelsBody = (await models.json()) as { models: { credential?: unknown }[] }; + expect(modelsBody.models).toHaveLength(1); + expect(modelsBody.models[0]!.credential).toBeTruthy(); + }); + + it("Agent 级隔离:不同 Agent 的 vault 互相独立;不存在的 Agent 404", async () => { + await owner.put(vaultPath, { entries: [{ key: "ONLY_DEFAULT", value: "v-default-1" }] }); + + // Another Agent in the same Project: empty table, and writes don't affect each other. + expect( + (await owner.post(`/api/projects/${projectId}/agents`, { agentId: "other_agent" })).status, + ).toBe(201); + const otherPath = `/api/projects/${projectId}/agents/other_agent/vault`; + const empty = (await (await owner.get(otherPath)).json()) as VaultResponse; + expect(empty.entries).toEqual([]); + await owner.put(otherPath, { entries: [{ key: "ONLY_OTHER", value: "v-other-1" }] }); + const def = (await (await owner.get(vaultPath)).json()) as VaultResponse; + expect(def.entries.map((e) => e.key)).toEqual(["ONLY_DEFAULT"]); + + // Agent doesn't exist (including traversal-style ids blocked by requireValidId) → 404. + expect((await owner.get(`/api/projects/${projectId}/agents/no-such-agent/vault`)).status).toBe( + 404, + ); + expect( + ( + await owner.put(`/api/projects/${projectId}/agents/no-such-agent/vault`, { + entries: [{ key: "K", value: "v" }], + }) + ).status, + ).toBe(404); + }); + + it("键名与请求体形状校验 400", async () => { + const cases: unknown[] = [ + { entries: [{ key: "1BAD", value: "v" }] }, // starts with a digit + { entries: [{ key: "BAD-DASH", value: "v" }] }, // hyphen + { entries: [{ key: "BAD KEY", value: "v" }] }, // space + { entries: [{ key: "OK_KEY", value: "" }] }, // empty value + { + entries: [ + { key: "DUP", value: "a" }, + { key: "DUP", value: "b" }, + ], + }, // duplicate key + { entries: [{ key: "BIG", value: "x".repeat(8193) }] }, // value too long (>8192, guards against exec E2BIG) + { entries: "nope" }, // entries is not an array + { entries: [42] }, // entry is not an object + ]; + for (const body of cases) { + expect((await owner.put(vaultPath, body)).status).toBe(400); + } + }); +}); diff --git a/packages/server/test/workspace-files.test.ts b/packages/server/test/workspace-files.test.ts new file mode 100644 index 0000000..a1827ca --- /dev/null +++ b/packages/server/test/workspace-files.test.ts @@ -0,0 +1,224 @@ +/** + * Unit tests for the Workspace files service: directory-listing order, read/write, + * path confinement (`..` traversal and symlink escape), size-limit protection, + * batch existence checks (files/stat); and the Agent delete route (default_agent + * cannot be deleted, owner-only, directory and index cleanup). + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { WorkspaceFilesService } from "../src/services/workspace-files-service.js"; +import type { + AgentCreateResponse, + ProjectCreateResponse, + SessionCreateResponse, +} from "../src/api/types.js"; +import { apiClient, createTestApp, makeTempRoot, provisionUser } from "./helpers.js"; +import type { TestApp } from "./helpers.js"; + +describe("workspace-files-service", () => { + let ws: string; + let outside: string; + const svc = new WorkspaceFilesService(); + + beforeEach(async () => { + ws = await makeTempRoot(); + outside = await makeTempRoot(); + await fs.mkdir(path.join(ws, "sub")); + await fs.writeFile(path.join(ws, "b.txt"), "hello"); + await fs.writeFile(path.join(ws, "sub", "c.md"), "# md"); + await fs.writeFile(path.join(outside, "secret.txt"), "secret"); + }); + afterEach(async () => { + await fs.rm(ws, { recursive: true, force: true }); + await fs.rm(outside, { recursive: true, force: true }); + }); + + it("列目录:dir 在前、按名称排序;子目录路径生效", async () => { + const root = await svc.list(ws, ""); + expect(root.entries.map((e) => `${e.kind}:${e.name}`)).toEqual(["dir:sub", "file:b.txt"]); + const sub = await svc.list(ws, "sub"); + expect(sub.entries.map((e) => e.name)).toEqual(["c.md"]); + }); + + it("读文件:内容与 content-type;目录/不存在报错", async () => { + const file = await svc.read(ws, "sub/c.md"); + expect(file.data.toString()).toBe("# md"); + expect(file.contentType).toContain("markdown"); + await expect(svc.read(ws, "sub")).rejects.toMatchObject({ status: 400 }); + await expect(svc.read(ws, "nope.txt")).rejects.toMatchObject({ status: 404 }); + }); + + it("写文件:覆盖写入;父目录缺失时自动补建(上传文件夹保留目录结构)", async () => { + await svc.write(ws, "sub/new.txt", Buffer.from("data")); + expect(await fs.readFile(path.join(ws, "sub", "new.txt"), "utf8")).toBe("data"); + await svc.write(ws, "missing/deep/x.txt", Buffer.from("d")); + expect(await fs.readFile(path.join(ws, "missing", "deep", "x.txt"), "utf8")).toBe("d"); + }); + + it("路径限域:`..` 穿越与绝对路径均拒绝", async () => { + await expect(svc.list(ws, "../")).rejects.toMatchObject({ status: 400 }); + await expect(svc.read(ws, `../${path.basename(outside)}/secret.txt`)).rejects.toMatchObject({ + status: 400, + }); + await expect(svc.write(ws, "../escape.txt", Buffer.from("x"))).rejects.toMatchObject({ + status: 400, + }); + }); + + it("Workspace 为文件系统根:子目录可正常下钻(前缀拼接 '//' 回归)", async () => { + const root = path.parse(ws).root; + const sub = await svc.list(root, path.relative(root, path.join(ws, "sub"))); + expect(sub.entries.map((e) => e.name)).toEqual(["c.md"]); + }); + + it("指向 Workspace 内目录的符号链接:kind 为 dir 且可下钻", async () => { + await fs.symlink(path.join(ws, "sub"), path.join(ws, "link-sub")); + const root = await svc.list(ws, ""); + expect(root.entries.map((e) => `${e.kind}:${e.name}`)).toEqual([ + "dir:link-sub", + "dir:sub", + "file:b.txt", + ]); + const viaLink = await svc.list(ws, "link-sub"); + expect(viaLink.entries.map((e) => e.name)).toEqual(["c.md"]); + }); + + it("符号链接逃逸:链接指向 Workspace 外时读写均拒绝", async () => { + await fs.symlink(outside, path.join(ws, "link-out")); + await expect(svc.list(ws, "link-out")).rejects.toMatchObject({ status: 400 }); + await expect(svc.read(ws, "link-out/secret.txt")).rejects.toMatchObject({ status: 400 }); + // Writing outside via a directory symlink: caught by the parent-directory realpath check. + await expect(svc.write(ws, "link-out/evil.txt", Buffer.from("x"))).rejects.toMatchObject({ + status: 400, + }); + // Auto-creation under a missing path is equally restricted: if the nearest + // existing ancestor is a symlink pointing outside, mkdir must not be used to escape. + await expect(svc.write(ws, "link-out/new/evil.txt", Buffer.from("x"))).rejects.toMatchObject({ + status: 400, + }); + expect( + await fs + .stat(path.join(outside, "new")) + .then(() => true) + .catch(() => false), + ).toBe(false); + }); + + it("末段符号链接写入:O_NOFOLLOW 拒绝借刀覆盖域外文件", async () => { + // The Agent has a symlink inside the Workspace pointing to an outside file; an upload attempts to overwrite it. + const victim = path.join(outside, "secret.txt"); + await fs.symlink(victim, path.join(ws, "report.pdf")); + await expect(svc.write(ws, "report.pdf", Buffer.from("PWNED"))).rejects.toMatchObject({ + status: 400, + }); + // The outside file's content is unchanged. + expect(await fs.readFile(victim, "utf8")).toBe("secret"); + }); +}); + +describe("files/stat 路由(批量存在性检查)", () => { + let t: TestApp; + let owner: ReturnType; + let outsider: ReturnType; + let sessionId: string; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner"); + const b = await provisionUser(t.app, "outsider"); + owner = apiClient(t.app, a.cookie); + outsider = apiClient(t.app, b.cookie); + const created = (await ( + await owner.post("/api/projects", { projectId: "owner-stat", name: "项目" }) + ).json()) as ProjectCreateResponse; + const projectId = created.project.projectId; + await owner.put(`/api/projects/${projectId}/models`, { + defaultModel: { provider: "anthropic", modelId: "claude-sonnet-4-6" }, + models: [{ provider: "anthropic", modelId: "claude-sonnet-4-6", contextWindow: 128000 }], + }); + const sess = (await ( + await owner.post(`/api/projects/${projectId}/agents/default_agent/sessions`, {}) + ).json()) as SessionCreateResponse; + sessionId = sess.session.sessionId; + await fs.mkdir(path.join(sess.session.workspace, "sub")); + await fs.writeFile(path.join(sess.session.workspace, "a.txt"), "A"); + await fs.writeFile(path.join(sess.session.workspace, "sub", "b.md"), "B"); + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("存在的文件保序去重返回;不存在 / 目录 / 越界一律按不存在计且恒 200", async () => { + const res = await owner.post(`/api/sessions/${sessionId}/files/stat`, { + paths: ["sub/b.md", "a.txt", "sub/b.md", "nope.txt", "sub", "../escape.txt", "/etc/passwd"], + }); + expect(res.status).toBe(200); + expect(await res.json()).toEqual({ existing: ["sub/b.md", "a.txt"] }); + + const empty = await owner.post(`/api/sessions/${sessionId}/files/stat`, { paths: [] }); + expect(empty.status).toBe(200); + expect(await empty.json()).toEqual({ existing: [] }); + }); + + it("非法 body → 400:非数组 / 非字符串项 / 超量 / 超长", async () => { + const url = `/api/sessions/${sessionId}/files/stat`; + expect((await owner.post(url, { paths: "a.txt" })).status).toBe(400); + expect((await owner.post(url, { paths: [1] })).status).toBe(400); + const tooMany = Array.from({ length: 101 }, () => "a.txt"); + expect((await owner.post(url, { paths: tooMany })).status).toBe(400); + expect((await owner.post(url, { paths: ["x".repeat(513)] })).status).toBe(400); + }); + + it("外人访问 → 404(不泄露存在性)", async () => { + const res = await outsider.post(`/api/sessions/${sessionId}/files/stat`, { + paths: ["a.txt"], + }); + expect(res.status).toBe(404); + }); +}); + +describe("agent 删除路由", () => { + let t: TestApp; + let owner: ReturnType; + let outsider: ReturnType; + let projectId: string; + + beforeEach(async () => { + t = await createTestApp(); + const a = await provisionUser(t.app, "owner"); + const b = await provisionUser(t.app, "outsider"); + owner = apiClient(t.app, a.cookie); + outsider = apiClient(t.app, b.cookie); + const created = (await ( + await owner.post("/api/projects", { projectId: "owner-ws", name: "项目" }) + ).json()) as ProjectCreateResponse; + projectId = created.project.projectId; + }); + afterEach(async () => { + await t.cleanup(); + }); + + it("owner 删除 Agent:204,目录与列表项消失;default_agent 409;外人 404", async () => { + const created = (await ( + await owner.post(`/api/projects/${projectId}/agents`, { agentId: "temp_agent", name: "临时" }) + ).json()) as AgentCreateResponse; + const agentId = created.agent.agentId; + const dir = path.join(t.root, projectId, "agents", agentId); + await fs.access(dir); // directory exists after creation + + const outsiderRes = await outsider.delete(`/api/projects/${projectId}/agents/${agentId}`); + expect(outsiderRes.status).toBe(404); // no access → don't leak existence + + const res = await owner.delete(`/api/projects/${projectId}/agents/${agentId}`); + expect(res.status).toBe(204); + await expect(fs.access(dir)).rejects.toThrow(); + const list = (await (await owner.get(`/api/projects/${projectId}/agents`)).json()) as { + agents: Array<{ agentId: string }>; + }; + expect(list.agents.some((x) => x.agentId === agentId)).toBe(false); + + const def = await owner.delete(`/api/projects/${projectId}/agents/default_agent`); + expect(def.status).toBe(409); + }); +}); diff --git a/packages/server/test/workspace-guard.test.ts b/packages/server/test/workspace-guard.test.ts new file mode 100644 index 0000000..ae44851 --- /dev/null +++ b/packages/server/test/workspace-guard.test.ts @@ -0,0 +1,61 @@ +/** + * Unit tests for Workspace validation: only requires an existing directory; + * location is not constrained to the Project directory (reachability is + * governed by file permissions). + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { HttpError } from "../src/http/errors.js"; +import { assertWorkspaceAllowed } from "../src/services/workspace-guard.js"; +import { makeTempRoot } from "./helpers.js"; + +describe("workspace-guard", () => { + let root: string; + let projectA: string; + + beforeEach(async () => { + root = await makeTempRoot(); + projectA = path.join(root, "project-aaaa0001"); + await fs.mkdir(path.join(projectA, "workdir"), { recursive: true }); + }); + afterEach(async () => { + await fs.rm(root, { recursive: true, force: true }); + }); + + const guard = (workspace: string) => assertWorkspaceAllowed({ workspace }); + + it("已存在的目录放行并返回 realpath", async () => { + const ws = await guard(path.join(projectA, "workdir")); + expect(ws).toBe(await fs.realpath(path.join(projectA, "workdir"))); + }); + + it("Project 目录之外的任意目录同样放行", async () => { + const outside = path.join(root, "elsewhere"); + await fs.mkdir(outside, { recursive: true }); + await expect(guard(outside)).resolves.toBe(await fs.realpath(outside)); + }); + + it("符号链接解析为 realpath 后返回", async () => { + const outside = path.join(root, "linked"); + await fs.mkdir(outside, { recursive: true }); + const link = path.join(projectA, "escape"); + await fs.symlink(outside, link, "dir"); + await expect(guard(link)).resolves.toBe(await fs.realpath(outside)); + }); + + it("不存在的路径 → 400 workspace_not_found", async () => { + const err = await guard(path.join(projectA, "ghost")).catch((e: unknown) => e); + expect(err).toBeInstanceOf(HttpError); + expect((err as HttpError).status).toBe(400); + expect((err as HttpError).code).toBe("workspace_not_found"); + }); + + it("文件(非目录)→ 400 workspace_not_found", async () => { + const file = path.join(projectA, "a-file.txt"); + await fs.writeFile(file, "x", "utf8"); + const err = await guard(file).catch((e: unknown) => e); + expect((err as HttpError).status).toBe(400); + expect((err as HttpError).code).toBe("workspace_not_found"); + }); +}); diff --git a/packages/server/tsconfig.json b/packages/server/tsconfig.json new file mode 100644 index 0000000..8cd1715 --- /dev/null +++ b/packages/server/tsconfig.json @@ -0,0 +1,7 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { + "rootDir": "." + }, + "include": ["src", "test"] +} diff --git a/packages/server/tsup.config.ts b/packages/server/tsup.config.ts new file mode 100644 index 0000000..9267bed --- /dev/null +++ b/packages/server/tsup.config.ts @@ -0,0 +1,11 @@ +import { defineConfig } from "tsup"; + +export default defineConfig({ + // Explicitly name entries to preserve the dist/api/types.js subpath (exports "./api" points to it). + entry: { index: "src/index.ts", "api/types": "src/api/types.ts" }, + format: ["esm"], + target: "node22", + dts: true, + clean: true, + sourcemap: true, +}); diff --git a/packages/skills/README.md b/packages/skills/README.md new file mode 100644 index 0000000..c1e65ee --- /dev/null +++ b/packages/skills/README.md @@ -0,0 +1,32 @@ +# @prismshadow/penguin-skills + +The PenguinHarness built-in skill library. A Skill is a directory with a `SKILL.md` (frontmatter: name, description, version, updated) — files are the runtime source of truth and ship raw in this package's npm tarball. + +Skills follow the "index first, body on demand" design: only their metadata is injected into an Agent's system prompt; the Agent reads the full `SKILL.md` via shell when it actually needs it. + +Included skills: + +| Group | Skills | +| --- | --- | +| Agent Development | `agent-creation`, `benchmark-design`, `agent-evaluation`, `agent-optimization` | +| Data Analysis | `data-analysis` | +| Penguin Development | `penguin-sdk`, `penguin-cli`, `agenthub-models` | +| Web Development | `web-design` | +| Software Engineering | `software-engineering` | + +The first group powers the self-improvement loop: design a Benchmark, evaluate the Target Agent, optimize it to version N+1 with a snapshot before every round. + +## Documentation + +- [Skills](https://prism-shadow.github.io/penguin-harness/docs/skills) +- [Self-Improvement](https://prism-shadow.github.io/penguin-harness/docs/self-improvement) + +## Development + +```bash +pnpm --filter @prismshadow/penguin-skills build # tsup → dist/ (loader API) +pnpm --filter @prismshadow/penguin-skills typecheck +pnpm --filter @prismshadow/penguin-skills test +``` + +Part of [PenguinHarness](https://github.com/Prism-Shadow/penguin-harness) · Apache-2.0 diff --git a/packages/skills/package.json b/packages/skills/package.json new file mode 100644 index 0000000..ff3ce3d --- /dev/null +++ b/packages/skills/package.json @@ -0,0 +1,39 @@ +{ + "name": "@prismshadow/penguin-skills", + "version": "0.0.1", + "type": "module", + "description": "PenguinHarness skill library: built-in SKILL.md documents and skill groups, decoupled from core.", + "license": "Apache-2.0", + "repository": { + "type": "git", + "url": "git+https://github.com/Prism-Shadow/penguin-harness.git", + "directory": "packages/skills" + }, + "exports": { + ".": { + "types": "./dist/index.d.ts", + "import": "./dist/index.js" + } + }, + "main": "./dist/index.js", + "types": "./dist/index.d.ts", + "scripts": { + "typecheck": "tsc --noEmit -p tsconfig.json", + "test": "vitest run --passWithNoTests", + "build": "tsup" + }, + "devDependencies": { + "@types/node": "^24.0.0", + "tsup": "^8.3.0", + "typescript": "^5.6.0", + "vitest": "^2.1.0" + }, + "files": [ + "dist", + "skills", + "LICENSE" + ], + "publishConfig": { + "access": "public" + } +} diff --git a/packages/skills/skills/agent-creation/SKILL.md b/packages/skills/skills/agent-creation/SKILL.md new file mode 100644 index 0000000..7b0eb68 --- /dev/null +++ b/packages/skills/skills/agent-creation/SKILL.md @@ -0,0 +1,73 @@ +--- +name: agent-creation +description: Turn a user requirement into a concrete agent — write the target agent's AGENTS.md and install the skills it needs. +short_description: Turn a requirement into a working agent. +short_description_zh: 把需求变成可用的 Agent。 +version: 1 +updated: 2026-07-17T00:00:00Z +--- + +# Agent Creation + +This skill turns a user requirement into a working agent configuration — plain files in the target agent's directory. + +## Before you start + +If the user's message only invokes this skill (e.g. "use agent-creation skill") without a concrete requirement, ask the user what agent they want and what it should do. Do not start until the requirement is clear. + +## Locate the target agent + +All agents of this project live side by side in the project directory: + +```bash +PROJECT_DIR="" # the Project Dir value from your Environment section +ls "$PROJECT_DIR/agents" # existing agents (each is a folder here) +TARGET="$PROJECT_DIR/agents/" # the agent to configure +``` + +An agent directory contains `agent_state/` (`system_config.yaml`, `AGENTS.md`, `skills/`, `memory/`, `tools/`) plus `scratchpad/` — and `traces/`, which appears once the agent has run at least once. + +## Write AGENTS.md + +`agent_state/AGENTS.md` is injected into the agent's system prompt — it is where the user requirement becomes behavior. Keep `system_config.yaml`'s `system_prompt` untouched (that is the stable system layer); put everything requirement-specific in AGENTS.md: + +- Role — what the agent is for, in one or two sentences. +- Domain guidance — the concrete rules, steps and constraints derived from the user requirement. + +Be concise: AGENTS.md is prompt context, not documentation. + +## Install skills + +A skill is a directory `agent_state/skills//` containing a `SKILL.md`: + +```md +--- +name: +description: +version: 1 +updated: +--- + + +``` + +The frontmatter may also carry optional `short_description` and `short_description_zh` lines (a short UI blurb and its Chinese variant) — the UI prefers them for display, while prompt injection always uses the English `description`. + +Installing is all it takes: the frontmatter metadata of every `SKILL.md` under `skills/` is injected into the target agent's system prompt automatically — do not register skills in AGENTS.md. + +Write skills yourself, or fetch existing ones from the internet with shell commands (`curl`, `git clone`) and place them under `skills/`. Anything fetched from the internet must be read in full and reviewed before installing — a skill becomes durable instructions the target agent will follow in every future session; never install one you have not read, and tell the user what it does. + +## Set name and description + +In the target's `agent_state/system_config.yaml`, set the top-level `name:` and `description:` fields so the agent is recognizable in lists. Edit only these two fields. + +## Creating a brand-new agent + +Prefer configuring an agent the user already created. If you must create one from scratch: pick a short id (letters, digits, `_`, `-`), copy the default agent's `system_config.yaml` as the base, and create the layout described above: + +```bash +mkdir -p "$TARGET/agent_state/skills" "$TARGET/agent_state/memory" "$TARGET/agent_state/tools" "$TARGET/scratchpad" +cp "$PROJECT_DIR/agents/default_agent/agent_state/system_config.yaml" "$TARGET/agent_state/" +``` + +A new agent starts with no skills — install only what it needs. Then write its AGENTS.md, name and description as above. diff --git a/packages/skills/skills/agent-creation/icon.svg b/packages/skills/skills/agent-creation/icon.svg new file mode 100644 index 0000000..22e4bbe --- /dev/null +++ b/packages/skills/skills/agent-creation/icon.svg @@ -0,0 +1,6 @@ + + + + + + diff --git a/packages/skills/skills/agent-evaluation/SKILL.md b/packages/skills/skills/agent-evaluation/SKILL.md new file mode 100644 index 0000000..1b228c8 --- /dev/null +++ b/packages/skills/skills/agent-evaluation/SKILL.md @@ -0,0 +1,123 @@ +--- +name: agent-evaluation +description: Run and score exactly one Benchmark Case run with CLI execution, Trace provenance checks, and private Rubric isolation. +short_description: Run and score one isolated Benchmark Case. +short_description_zh: 隔离执行并评分一个 Benchmark Case。 +version: 1 +updated: 2026-07-17T17:08:17Z +--- + +# Agent Evaluation + +Act as an internal leaf worker. For one valid request, run and score exactly one Benchmark Case once, then return minimal protocol metadata. Do not design or refine the Benchmark, modify the Test Agent State, or write `scoreboard.yaml`. Do not use `run_subagent` or `input_subagent`. + +## Before you start + +This Skill is invoked by `benchmark-design` or Benchmark mode in `agent-optimization`. Require one unambiguous protocol request containing every identity field below. If the request is missing, duplicated, or conflicting, return `invalid_request` without creating a Workspace or launching the Test Agent. Do not ask an interactive clarification from this leaf worker. + +## Privacy boundary + +Before the final protocol YAML, emit no assistant text; use private reasoning and tool calls only. Never serialize Statement or artifact contents, Rubric items, expected values, correct outcomes, per-item scoring, diagnostics, secret configuration, Workspace paths, or Trace paths into an assistant message. The final assistant message is the protocol YAML only. It may echo the public identity fields supplied by the caller. + +A valid request contains exactly one value for each field: + +```text +protocol_version: 1 +case_id: +run: <1_based_run_index> +expected_version: +test_agent_id: +benchmark_dir: +provider: +model_id: +``` + +## Validate and prepare + +Resolve the Project, Test Agent, Benchmark, and Case only from the explicit request and Environment Project Dir. Reject traversal, symlink escape, or any path outside the requested Test Agent. Never read a Project configuration file, credential, or vault. + +Require: + +```text +/agent_state/system_config.yaml +/benchmark_config.toml +//statement/README.md +//rubric/README.md +``` + +Require `benchmark_config.toml` to contain a positive integer `runs`; the requested `run` must be within `1..runs`. The canonical State version is the top-level `version` in `system_config.yaml`, defaulting to 1, and must equal `expected_version`. + +Read and retain the exact Statement and Rubric bytes before launch. Reject an unusable, contradictory, non-atomic, or unbounded Rubric. The Rubric must declare a finite Case maximum; the returned score must fall within `0..case_max`. + +Create a collision-checked Workspace at `/workspaces/tmp-<8hex>`. Copy only the contents of `statement/` into it. Never copy, link, or disclose `rubric/`, and never reuse another Case or run's Workspace. + +## Launch and bind the Test Session + +Use an existing verified Penguin CLI or repository-local launcher already available in the runtime. Do not install a CLI and do not use `penguin run` as a probe. If no launcher is available, return `cli_failed`. + +Run the Test Agent exactly once in the foreground with a fresh top-level Session: + +```bash +PROJECT_DIR="" +PROJECT_ID="$(basename "$PROJECT_DIR")" +PENGUIN_HOME="$(dirname "$PROJECT_DIR")" +WORKSPACE="$PROJECT_DIR/agents//workspaces/" +export PENGUIN_HOME +penguin run --message "Read README.md in the current Workspace and complete the task exactly as specified there." \ + --provider "" --model-id "" --project-id "$PROJECT_ID" \ + --agent-id "" --workspace "$WORKSPACE" --approve allow-all +``` + +Use the exact Project, Test Agent, Model pair, and Workspace. Do not fall back to another value. Poll the same process until it exits. A nonzero, interrupted, or misrouted launch is `cli_failed`, not score zero. Do not relaunch within the same run or target processes by a global name or pattern. + +Read the canonical State version before and after the Test run; any change is `version_changed`. Confirm the Statement and Rubric bytes are unchanged before scoring. + +Search only the explicitly requested Test Agent's `traces/` tree and never inspect another Agent's traces. Group rotated shards by Session. Evaluate every Session group that could contain the exact Workspace match; never infer ownership from recency or a fixed-size latest subset. Bind the Test Trace mechanically from `session_meta`: + +- `payload.workspace` equals the unique Workspace; +- `payload.agent_state` equals the exact Test Agent State path; +- `payload.provider` equals the requested provider; +- `payload.model_id` equals the requested model id. + +When matching Test subagents exist, exclude child ids referenced by subagent events and require one unique matching root Test Session. Unrelated concurrent traces are not conflicts. Missing, multiple, malformed, or identity-mismatched roots are `provenance_mismatch`. + +## Score and account + +Inspect only the unique Test Workspace, its bound Test Trace, and the retained private Rubric. Apply every atomic item exactly and normalize only allowed equivalents. A missing, malformed, wrong-type, or incorrect Test artifact is ordinary scored Test Agent behavior: apply the Rubric's zero or partial credit and return `status: ok`. Only a changed or unusable Rubric, or a non-finite/out-of-range result, is `invalid_score`. Detailed reasoning remains in Evaluator Trace. + +Compute `duration_ms` from the bound root Test Session, not from the Evaluator. Compute cost only from that root and child Sessions mechanically referenced by subagent events whose traces are available within the explicitly requested Test Agent's `traces/` tree. For each included Session, use final cumulative token usage rather than summing intermediate cumulative events, and apply the matching public `(provider, model_id)` pricing. If any referenced child trace or usage is unavailable there, including because the Test Session delegated to another Agent, or if any included usage is unpriced, return `cost: null` rather than inspecting another Agent or reporting a known partial sum as complete cost. + +## Return protocol + +Emit exactly one plain YAML document beginning with `protocol_version:` and stop. Do not use a code fence or add explanations. + +On success: + +```text +protocol_version: 1 +status: ok +case_id: +run: +expected_version: +provider: +model_id: +score: <0_to_case_max> +cost: +duration_ms: +session_id: +``` + +On failure: + +```text +protocol_version: 1 +status: infrastructure_failure +case_id: +run: +expected_version: +provider: +model_id: +failure_code: +``` + +Stable codes are `invalid_request`, `invalid_statement`, `invalid_rubric`, `cli_failed`, `provenance_mismatch`, `version_changed`, and `invalid_score`. Do not include score, cost, duration, Session id, private data, or optimization advice on failure. diff --git a/packages/skills/skills/agent-evaluation/icon.svg b/packages/skills/skills/agent-evaluation/icon.svg new file mode 100644 index 0000000..28ae6b5 --- /dev/null +++ b/packages/skills/skills/agent-evaluation/icon.svg @@ -0,0 +1,6 @@ + + + + + + diff --git a/packages/skills/skills/agent-optimization/SKILL.md b/packages/skills/skills/agent-optimization/SKILL.md new file mode 100644 index 0000000..7d75970 --- /dev/null +++ b/packages/skills/skills/agent-optimization/SKILL.md @@ -0,0 +1,146 @@ +--- +name: agent-optimization +description: Improve an Agent State from direct feedback or versioned multi-Case Benchmark scores and score-linked Traces. +short_description: Improve an Agent from feedback or measured Benchmark results. +short_description_zh: 根据反馈或 Benchmark 结果改进 Agent。 +version: 1 +updated: 2026-07-17T17:08:17Z +--- + +# Agent Optimization + +Improve an existing Agent State. Use one-shot feedback mode for a direct correction, or Benchmark optimization mode for a measured loop. Do not mix their execution paths. One-shot mode does not require Agent Evaluator. Benchmark mode evaluates every configured Case run through Agent Evaluator and never launches or scores the Test Agent directly. + +## Before you start + +If the request supplies neither concrete feedback nor an explicit Test Agent and Benchmark, ask what to improve. Determine the mode before editing anything. + +Benchmark mode requires a top-level Session with `run_subagent`, a complete baseline series in `scoreboard.yaml`, and `agent-evaluation` installed on the current Agent. If any requirement is missing, stop and explain what the user must provide or install. Do not begin a partial optimization round. + +## Pick the target Agent + +A one-shot request normally names the target. A delegated request begins with `Caller agent: `, and an @-mention handoff contains ``. When one-shot mode has no explicit target, use that caller or origin; if neither exists, ask. Benchmark mode always requires an explicit Test Agent and Benchmark. + +Resolve paths from the Environment's Project Dir without recursively discovering the Project: + +```text +PROJECT_DIR = +PROJECT_ID = +PENGUIN_HOME = +TARGET = /agents/ +STATE = /agent_state +TRACES = /traces +BENCHMARK = /benchmarks/ +SCOREBOARD = /scoreboard.yaml +``` + +Never read a Project configuration file, credential, vault, private Rubric, Agent Evaluator State, Evaluator Workspace, or Evaluator Trace. + +## State editing policy + +Make the smallest complete edit supported by evidence and preserve unrelated instructions and files. + +- Behavioral, workflow, role, or domain guidance belongs in `agent_state/AGENTS.md` unless a relevant target-owned Skill already owns that reusable capability. +- Update a relevant target-owned `SKILL.md` when the behavior is a reusable capability shared across tasks. +- Create a narrowly named Skill only when the capability is reusable and no suitable Skill exists. Installing its directory is sufficient; do not register Skill metadata in AGENTS.md. +- Runtime limits belong in safe `system_config.yaml` fields. Do not edit `system_prompt` unless the user explicitly asks. +- Never modify a library-provided Skill such as `penguin-sdk` to carry target-specific behavior. + +Every edit must generalize beyond the observed run. Do not encode Case ids, exact expected outputs, Benchmark-specific constants, private criteria, or a guessed answer. Keep a recovered mapping, threshold, or formula only when repeated public evidence supports it as a durable rule; otherwise encode the reasoning and validation method. + +## Version and rollback discipline + +Before changing Agent State, read its canonical top-level `version` from `system_config.yaml`, defaulting to 1. The system owns Agent State snapshot archives and exposes them through Web export and import. Do not create, import, extract, or replace snapshot archives yourself. Require `/snapshots/v.tar.gz` to exist; if it is missing, stop and ask the user to export the current Agent State from Agent settings before continuing. + +For each file the edit will change, record its exact original bytes and whether it existed in a temporary directory outside `STATE`. Never include `.vault.toml`, an unrelated State file, or a snapshot archive. Write each candidate file through a temporary sibling, validate it, and rename it into place. Set the State version to `current + 1` exactly once before evaluation. + +If the candidate is rejected or a valid comparison cannot complete, restore only those recorded files, remove files created by the candidate, and verify that the prior version is active. If the active version or a candidate-owned file no longer matches the value written by this round, treat it as a concurrent mutation and stop without overwriting it. If rollback cannot be verified, stop and ask the user to restore the system snapshot through Web import. + +## One-shot feedback mode + +Use the user's feedback and relevant recent Trace when available. Turn that evidence into the smallest targeted change: + +- behavior, workflow, role, or domain guidance → update AGENTS.md or the relevant target-owned Skill; +- a missing reusable capability → install or update a narrowly scoped Skill; +- a runtime limit → adjust the relevant `system_config.yaml` field. + +Require the current-version system snapshot and record the exact originals of files being changed before editing. Increment the version once after the complete edit. If the edit cannot complete, roll back only those files. Report the evidence, changed State surface, new version, and reason. Do not claim measured improvement because this mode has no Benchmark comparison. + +## Benchmark optimization mode + +Require `benchmark_config.toml` with a positive integer `runs` and a complete baseline evaluation. The baseline Case set is frozen for optimization. Each reference or candidate evaluation must include exactly `runs` uniquely numbered runs for every frozen Case. + +Use the reference evaluation's `(provider, model_id)` pair unchanged so scores remain comparable. If the user wants a different Model, stop and ask for a new baseline series rather than comparing across Models. + +Before any candidate edit, read the canonical top-level State version and select a reference evaluation in the existing `(provider, model_id)` series. The selected reference is valid only when its `version` equals the active State version and its Cases and runs form the complete frozen Case × `runs` matrix. Never compare a candidate against an older-version, incomplete, or mixed-Model evaluation. + +If the active State has no such evaluation, measure it before optimizing: leave State unchanged, run the complete frozen matrix with the existing series' exact `(provider, model_id)` pair, and validate the results under the same rules used for a candidate. Retain the exact Scoreboard bytes and active State version before dispatch; append the completed no-edit reference through a temporary sibling, YAML validation, and atomic rename only if both remain unchanged. This appended evaluation becomes the reference. If any cell remains invalid after the allowed retry, provenance does not match, State or Scoreboard changes, or the atomic append cannot complete, stop before editing Agent State. + +You may inspect the complete target `agent_state/`, public Case Statements, the Scoreboard, and all Test traces referenced by the Scoreboard runs. Use only those explicit Case and Session ids. Never edit the Benchmark, Test traces, Project configuration, or another Agent. + +Scoreboard v2 uses this shape: + +```yaml +evaluations: + - time: "2026-07-17T00:00:00Z" + version: 2 + provider: deepseek + model_id: deepseek-v4-pro + summary_title: "Improved evidence validation" + summary: "Added a reusable validation step; all Cases improved without new instability." + score: 24 + cost: 0.04 + duration_ms: 60000 + cases: + - case: CASE-001-example + score: 24 + cost: 0.04 + duration_ms: 60000 + runs: + - score: 23 + cost: 0.03 + duration_ms: 58000 + session_id: session-1 + - score: 24 + cost: 0.04 + duration_ms: 60000 + session_id: session-2 + - score: 25 + cost: 0.05 + duration_ms: 62000 + session_id: session-3 +``` + +Case metrics are the means of valid `runs`; evaluation totals are the sums of Case means. Omit cost when any contributing run has unknown cost. `summary_title` and `summary` may describe public State changes, gains, regressions, and instability, but must not reveal private Rubric or Gold content, expected answers, or private scoring reasoning. + +## Optimization loop + +For each round: + +1. Reconfirm that the reference version equals the active top-level State version and that it contains the complete frozen Case × `runs` matrix. Then analyze its aggregate, Case scores, repeated runs, and every score-linked Test Trace. Use repeated runs to separate stable failure from variation; never select a convenient Trace. +2. State one falsifiable behavioral hypothesis connecting public evidence to a minimal State change. If no credible hypothesis remains, stop. +3. Confirm the current-version system snapshot exists, record the exact originals of every candidate-owned file, make the candidate edit, and set `version` to `current + 1` once. +4. Retain the exact Scoreboard bytes and candidate version. Build the complete Case-run matrix before dispatch. +5. Start one child per cell with `run_subagent`, and omit `agent_id` so the child reuses the current Agent. Each prompt begins with the caller identity and the sentence: Use the `agent-evaluation` Skill. Then it contains one request: + + ```text + Caller agent: + Use the `agent-evaluation` Skill. Return only its terminal protocol YAML. + protocol_version: 1 + case_id: + run: <1_based_run_index> + expected_version: + test_agent_id: + benchmark_dir: + provider: + model_id: + ``` + + For N Cases and R runs, emit all N × R independent calls in the same parallel tool-call group before waiting. Continue an active child through `input_subagent`; never duplicate it. +6. Parse only each child's last terminal `protocol_version: 1` YAML mapping. Keep every identity-matched `status: ok` result. Retry invalid cells once, dispatching all retries together with identical State, Model, Benchmark, and run identities. Never retry a valid scored cell. If any retry remains invalid, reject the round and roll back the candidate files. +7. Compute Case means and evaluation sums. Retain every score, cost when complete, duration, and Test Session id. Re-read State version and exact Scoreboard bytes, and verify that every candidate-owned file still matches the value written by this round. If State changed concurrently, stop without overwriting it. If only the Scoreboard changed, reject the round and roll back the candidate files. +8. Accept only a score strictly higher than the comparable reference evaluation. On improvement, keep the candidate State and append one evaluation through a temporary sibling, YAML validation, and atomic rename. On an equal or lower score, roll back the candidate files and append nothing. + +Each accepted round becomes the next reference. Stop when the user's target or round limit is met, no credible evidence-backed hypothesis remains, or infrastructure prevents another valid comparison. Do not mutate State as random search. + +At the end, report the accepted score curve, State versions and changes, rejected hypotheses and rollbacks, Test Session ids, stop reason, and limitations. Distinguish the active tested State from any unscored State; never claim a Scoreboard score applies to a later untested edit. diff --git a/packages/skills/skills/agent-optimization/icon.svg b/packages/skills/skills/agent-optimization/icon.svg new file mode 100644 index 0000000..a6a949d --- /dev/null +++ b/packages/skills/skills/agent-optimization/icon.svg @@ -0,0 +1,7 @@ + + + + + + + diff --git a/packages/skills/skills/agenthub-models/SKILL.md b/packages/skills/skills/agenthub-models/SKILL.md new file mode 100644 index 0000000..7ba0678 --- /dev/null +++ b/packages/skills/skills/agenthub-models/SKILL.md @@ -0,0 +1,123 @@ +--- +name: agenthub-models +description: Call model APIs through @prismshadow/agenthub — streaming text generation, image generation, speech synthesis and embeddings with one client. +short_description: Call model APIs with one AgentHub client. +short_description_zh: 用一个 AgentHub 客户端调用模型 API。 +version: 1 +updated: 2026-07-17T00:00:00Z +--- + +# AgentHub Model APIs + +`@prismshadow/agenthub` is a unified TypeScript client for model APIs: streaming text, image generation, speech synthesis and embeddings behind one entry point. + +```bash +npm install @prismshadow/agenthub +``` + +The only entry point is `AutoLLMClient`: + +```ts +import { AutoLLMClient } from "@prismshadow/agenthub"; + +const client = new AutoLLMClient({ model: "", apiKey: "", baseUrl: "", clientType: "" }); +``` + +`apiKey`, `baseUrl` and `clientType` are optional (see routing below). + +## Before you start + +If the user's message only invokes this skill (e.g. "use agenthub-models skill") without a concrete task, ask the user what they want to build. Do not write code until the requirement is clear. + +## Model IDs + +Use exact model ids. If an id is not in the table below and the user has not given one, ask the user to confirm the exact id before writing code. + +| Family | Official IDs | Gateway variants | +| ---------------- | --------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- | +| Gemini 3 | `gemini-3.1-pro-preview`, `gemini-3.5-flash`, `gemini-3.1-flash-lite` | — | +| Gemini 3 image | `gemini-3.1-flash-image-preview`, `gemini-3-pro-image-preview` | — | +| Gemini 3 TTS | `gemini-3.1-flash-tts-preview` | — | +| Gemini embedding | `gemini-embedding-2` | — | +| Claude | `claude-sonnet-4-6`, `claude-opus-4-7`, `claude-opus-4-8` | — | +| GPT | `gpt-5.4`, `gpt-5.4-mini`, `gpt-5.4-nano`, `gpt-5.5` | — | +| OpenAI embedding | `text-embedding-3-small`, `text-embedding-3-large` | — | +| Kimi K2.6 | `kimi-k2.6` | OpenRouter `moonshotai/kimi-k2.6`; SiliconFlow `Pro/moonshotai/Kimi-K2.6` | +| DeepSeek V4 | `deepseek-v4-pro`, `deepseek-v4-flash` | OpenRouter `deepseek/deepseek-v4-pro`, `deepseek/deepseek-v4-flash`; SiliconFlow `deepseek-ai/DeepSeek-V4-Pro`, `deepseek-ai/DeepSeek-V4-Flash` | +| GLM 5.1 | `glm-5.1` | OpenRouter `z-ai/glm-5.1`; SiliconFlow `Pro/zai-org/GLM-5.1` | + +Gateway model lists can be queried online: + +```bash +curl https://openrouter.ai/api/v1/models +curl --request GET --url https://api.siliconflow.cn/v1/models --header 'Authorization: Bearer ' +``` + +## Routing and credentials + +- Without `clientType`, the client auto-routes by model id substring: `gemini-3*`, `gemini-embedding`, `claude` 4-6/4-7/4-8, `gpt-5.4`/`gpt-5.5`, `glm-5`, `kimi-k2.5`/`kimi-k2.6`, `deepseek-v4`, `openai`+`embedding` (embeddings), `openai`. Ids matching none of these throw. The gateway variants in the table above hit the same substrings, so they route to the right family — just set `baseUrl` to the gateway endpoint. +- For any other OpenAI chat-completion compatible model (e.g. Qwen series via OpenRouter or SiliconFlow), pass `clientType: "openai"` plus `baseUrl` (embeddings endpoints use a different client type — see Embeddings below). +- API key: constructor parameter first, then the provider environment variable — `DEEPSEEK_API_KEY`, `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `GEMINI_API_KEY`, `ZAI_API_KEY`, `MOONSHOT_API_KEY`. Base URLs read the same names with `_BASE_URL`. + +## Streaming text + +```ts +for await (const event of client.streamingResponseStateful({ + message: { role: "user", content_items: [{ type: "text", text: "Hello" }] }, + config: {}, +})) { + for (const item of event.content_items) { + if (item.type === "text") process.stdout.write(item.text); + } +} +``` + +- Each `event` is a `UniEvent`: `event_type` is `start` | `delta` | `stop`, and `content_items` carry the increments. +- `config` accepts `max_tokens`, `temperature`, `system_prompt`, `thinking_level` (the `ThinkingLevel` enum, `NONE` to `XHIGH`) and `tools`. +- `streamingResponseStateful` keeps conversation history inside the client; manage it with `getHistory()` / `setHistory(history)` / `clearHistory()`. The stateless variant is `streamingResponse({ messages, config })`. + +## Image generation + +Use a Gemini image model (see Model IDs) and set `config.image_config` (optional `aspect_ratio`, and `image_size` of `"1K"` | `"2K"`): + +```ts +import fs from "node:fs"; + +const client = new AutoLLMClient({ model: "gemini-3.1-flash-image-preview" }); +for await (const event of client.streamingResponseStateful({ + message: { role: "user", content_items: [{ type: "text", text: "A penguin on a glacier" }] }, + config: { image_config: { aspect_ratio: "16:9", image_size: "2K" } }, +})) { + for (const item of event.content_items) { + if (item.type === "inline_data") fs.writeFileSync("image.png", item.data); + } +} +``` + +Images arrive as `inline_data` content items (`data` is a Buffer, with `mime_type`). + +## Speech synthesis + +Use a Gemini TTS model (`gemini-3.1-flash-tts-preview`) and set `config.tts_config`: + +```ts +config: { tts_config: [{ voice: "Kore" }] } +``` + +- One entry → single voice; two entries → multi-speaker, and each entry must also set `speaker`. +- The `inline_data` output is raw PCM (24kHz 16-bit mono) — wrap it in a WAV header yourself before saving as `.wav`. + +## Embeddings + +Two routes: + +- Gemini: a model whose id contains `gemini-embedding` auto-routes (`gemini-embedding-2`). +- Any OpenAI-compatible embeddings endpoint: pass `clientType: "openai-embedding"` (plus `baseUrl` and `apiKey` as needed) — ids like `text-embedding-3-small` / `text-embedding-3-large` match no auto-route substring and would throw without it. + +Optional `config.embedding_config`: + +```ts +config: { embedding_config: { dimensions: 768 } } +``` + +The output arrives as `embedding` content items (`embedding` is a number array). diff --git a/packages/skills/skills/agenthub-models/icon.svg b/packages/skills/skills/agenthub-models/icon.svg new file mode 100644 index 0000000..f026085 --- /dev/null +++ b/packages/skills/skills/agenthub-models/icon.svg @@ -0,0 +1,9 @@ + + + + + + + + + diff --git a/packages/skills/skills/benchmark-design/SKILL.md b/packages/skills/skills/benchmark-design/SKILL.md new file mode 100644 index 0000000..c7e7ed9 --- /dev/null +++ b/packages/skills/skills/benchmark-design/SKILL.md @@ -0,0 +1,156 @@ +--- +name: benchmark-design +description: Design and calibrate a multi-Case capability Benchmark with repeated independent evaluations and a traceable baseline. +short_description: Design and calibrate an Agent capability Benchmark. +short_description_zh: 设计并校准 Agent 能力评测 Benchmark。 +version: 1 +updated: 2026-07-17T17:08:17Z +--- + +# Benchmark Design + +Create and calibrate a multi-Case Benchmark that discovers a specified Test Agent's capability boundary. Own the public Statements, private Rubrics, Case set, configuration, and final baseline. Do not modify the Test Agent State, launch the Test Agent directly, or score a Case yourself. + +## Before you start + +Require a Test Agent and the capability to measure. If either is missing, ask the user. A Benchmark run also requires a top-level Session with `run_subagent` and the current Agent must have `agent-evaluation` installed. If the Skill is missing or this Session is already a subagent, stop and ask the user to install the Skill or start a top-level Session. Do not begin a partial Benchmark. + +## Boundaries + +Access only the explicit Test Agent and Benchmark paths. Do not inspect another Agent, Project configuration files, Agent Evaluator State, Evaluator Workspace, or Evaluator Trace. Consume only each Evaluator's terminal protocol response and the returned Test Session id. + +Use the Environment's Project Dir and the explicit Test Agent id: + +```text +PROJECT_DIR = +PROJECT_ID = +PENGUIN_HOME = +TEST_AGENT_DIR = /agents/ +BENCHMARK_DIR = /agents//benchmarks/ +SCOREBOARD = /scoreboard.yaml +``` + +Derive a semantic Benchmark id when the user does not supply one. Require `agent_state/system_config.yaml`; its top-level `version` is the canonical State version and defaults to 1 when absent. + +## Benchmark contract + +Use this structure: + +```text +/ +├── benchmark_config.toml +├── scoreboard.yaml +└── CASE--/ + ├── statement/ + │ └── README.md + └── rubric/ + └── README.md +``` + +Both README files are required; either directory may contain supporting files. `statement/` is the complete public task and evidence. `rubric/` is private scoring material. Never mention private criteria or paths in the Statement. + +Create `benchmark_config.toml` first. The Model is deliberately not stored here; each evaluation records the actual `(provider, model_id)` pair. + +```toml +title = "" +description = "" +runs = 3 +``` + +Use `runs = 3` unless the user explicitly requests another positive integer. Initialize the Scoreboard with: + +```yaml +evaluations: [] +``` + +Every accepted evaluation follows scoreboard v2: + +```yaml +evaluations: + - time: "2026-07-17T00:00:00Z" + version: 1 + provider: deepseek + model_id: deepseek-v4-pro + summary_title: "Calibrated baseline" + summary: "Abbreviated baseline schema for one Case with three independent runs." + score: 18 + cost: 0.04 + duration_ms: 60000 + cases: + - case: CASE-001-example + score: 18 + cost: 0.04 + duration_ms: 60000 + runs: + - score: 17 + cost: 0.03 + duration_ms: 58000 + session_id: session-1 + - score: 18 + cost: 0.04 + duration_ms: 60000 + session_id: session-2 + - score: 19 + cost: 0.05 + duration_ms: 62000 + session_id: session-3 +``` + +Case `score`, `cost`, and `duration_ms` are the means of their valid `runs`. Evaluation totals are the sums of the Case means. Omit a Case or evaluation `cost` when any contributing run has unknown cost; never treat unknown as zero. Rubric maxima across the complete Case set should total 100 points so the evaluation score remains interpretable on a 0–100 scale. + +Builder writes one final baseline for the current Benchmark definition. A material change to a Statement, Rubric, Case set, or `runs` invalidates prior results: clear `evaluations`, recalibrate, and write a new baseline. Rejected candidates and provisional matrices remain only in Builder Trace. Evaluator never writes the Scoreboard. + +The public `summary_title` and `summary` may describe the tested State, Case-level score patterns, and instability. They must not reveal Rubric or Gold content, expected answers, private scoring reasoning, mappings, thresholds, formulas, or rules. + +## Design and calibrate + +Before writing Cases, define the observable difference between an Agent that has the requested capability and one that does not. Each Case must make that capability causally necessary, not merely share its topic. + +Run this counterfactual before accepting a Case: could a competent executor without the target capability complete it by mechanically following the Statement? If yes, reject or redesign it. A self-contained Statement specifies the task, available evidence, and required artifact without disclosing the reasoning, mapping, or rule the capability is supposed to recover. The evidence must still make the answer inferable. + +Build several independent, realistic end-to-end Cases with distinct capability-relevant failure modes. Do not manufacture low scores through missing essential evidence, trivia, formatting traps, excessive workload, or unstable infrastructure. + +Freeze every Rubric before evaluation. Use atomic observable conditions, exact points, reasonable equivalence rules, and meaningful partial credit. Test the Rubric mentally against full, partial, missing, malformed, wrong-type, and extra output. Never execute Test Agent-produced code while scoring. + +Use valid evidence to calibrate toward a user-supplied target; otherwise aim near 60/100. Treat a near-ceiling candidate as uncalibrated when a credible structural refinement remains. Audit high scores for shortcuts or leakage and low scores for ambiguity, missing evidence, unrelated difficulty, or a defective Rubric. More items, steps, or workload alone are not structural refinement. + +## Select the evaluation Model + +Use a user-specified `(provider, model_id)` pair when supplied. Otherwise resolve the Project default with the supported CLI, never by reading the hidden Project configuration: + +```bash +penguin config model list --project-id "" --root "" +``` + +The row marked `*` is the default. Keep the same pair through all candidate matrices for this calibration. The pair selects the Test Agent's CLI Session, not the Builder or Evaluator runtime model. + +## Run the Case-run matrix + +Read and retain the exact State version, Scoreboard bytes, configured positive `runs`, selected Model pair, and complete valid Case set. Build every unique Case-run cell before dispatch. + +Start one child per cell with `run_subagent`, and omit `agent_id` so the child reuses the current Agent. Each prompt must begin with the caller identity and the sentence: Use the `agent-evaluation` Skill. Then provide exactly one request: + +```text +Caller agent: +Use the `agent-evaluation` Skill. Return only its terminal protocol YAML. +protocol_version: 1 +case_id: +run: <1_based_run_index> +expected_version: +test_agent_id: +benchmark_dir: +provider: +model_id: +``` + +For N Cases and R runs, emit all N × R independent `run_subagent` calls in the same parallel tool-call group before waiting for results. Continue an active child through `input_subagent`; never duplicate it. + +Parse only the last terminal `protocol_version: 1` YAML mapping. Keep every identity-matched `status: ok` result. Retry only invalid cells once, with all retry cells dispatched together and the same State version, Model, Benchmark, and run identities. Never retry a valid scored cell. If any retry remains invalid, abandon the matrix. + +For a complete matrix, calculate Case means and evaluation sums using scoreboard v2. Retain every run's score, cost when known, duration, and Test Session id. Re-read State version and exact Scoreboard bytes; abandon the result if either changed. + +Use each returned Test Session id to inspect the exact Test Trace and artifact. Analyze all repeated runs; disagreement is capability instability, not permission to select a convenient result. Accept a candidate only when the capability caused the scored difference, evidence was sufficient, the Rubric was sound, and useful headroom remains. + +When calibration stops after a complete valid matrix, write the final baseline to a temporary sibling, parse it as YAML, then atomically rename it over `scoreboard.yaml`. Include the sorted Case set and sorted runs, a real UTC ISO-8601 time, the tested version and Model pair, and a privacy-safe summary. A near-ceiling final score is still recorded, but if the refinement budget ends without an acceptable candidate, report `calibration_failed` rather than calling the Benchmark ready. Leave `evaluations` empty only when no complete valid matrix exists. + +Report the Benchmark path, aggregate and Case scores, Test Session ids, refinements, stop reason, and limitations. diff --git a/packages/skills/skills/benchmark-design/icon.svg b/packages/skills/skills/benchmark-design/icon.svg new file mode 100644 index 0000000..ca47360 --- /dev/null +++ b/packages/skills/skills/benchmark-design/icon.svg @@ -0,0 +1,7 @@ + + + + + + + diff --git a/packages/skills/skills/data-analysis/SKILL.md b/packages/skills/skills/data-analysis/SKILL.md new file mode 100644 index 0000000..a632eff --- /dev/null +++ b/packages/skills/skills/data-analysis/SKILL.md @@ -0,0 +1,39 @@ +--- +name: data-analysis +description: Complete data-analysis tasks with bounded evidence inspection, explicit answer-changing decisions, native artifact handling, and final output verification. +short_description: Complete data analysis with a bounded, verified method. +short_description_zh: 以有界、可验证的方法完成数据分析任务。 +version: 1 +updated: 2026-07-18T00:00:00Z +--- + +# Data Analysis + +## Before you start + +Require a concrete data-analysis task, its available inputs, and the requested deliverable, location, and format. If the task or required materials are missing or ambiguous in an answer-changing way, ask before calculating or creating outputs. Otherwise proceed without forcing a fixed analysis template. + +## Success criteria + +- Derive the final result from one committed method: the selected evidence, answer-changing decisions, transformations, calculations, and judgments. +- Produce every output the task asks for, in the requested location and format, and ensure it reflects the committed method. +- Ground each answer-changing choice in the task materials rather than in a merely plausible nearby match. +- Prefer a complete, simple, defensible deliverable over an elaborate or exhaustive analysis that risks not being delivered. + +## Constraints + +- Use bounded probes first for large, unfamiliar, or expensive-to-read inputs. Narrow the scope, cap output, or use a timeout before expanding. +- Before calculating or producing final outputs, identify the few decisions that can change the answer, such as inclusion or exclusion, matching, boundary choices, transformations, formulas, and ranking criteria. Adapt this check to the task; do not force a fixed analysis template. +- Compare materially plausible alternatives only when they would change the final output. Use the smallest comparison needed to resolve the choice from the task materials, then commit to a method. +- Once the evidence supports a defensible answer, create the requested output files promptly. Do not continue open-ended thinking, searching, or polishing over alternatives that would not change the delivered answer. +- Use intermediate files only when they help compute or verify the result. Preserve the requested output format and structure unless the task asks for a change. +- Satisfy the requested deliverable with the simplest sufficient artifact and implementation. Avoid optional structure, styling, helper code, or reimplementation that is not required by the task. +- Keep final delivery steps short and robust. Avoid putting a long report, large dataset, or large script into one fragile streamed command when the deliverable can be created by shorter steps. Create the required artifact first, then refine only if the required outputs already exist. +- For spreadsheet or Office deliverables, prefer the tool path that preserves the native artifact contract. If a task depends on Excel formula recalculation, data tables, workbook formatting, or Office export, first check whether native Excel automation such as `xlwings` is available before falling back to LibreOffice or a hand-rolled model. Record formulas or inputs that will be temporarily overwritten, restore them before finalizing, and verify that the requested workbook, deck, document, or PDF can be opened and contains the expected tables, sheets, slides, or sections. Do not dump or reimplement an entire workbook when bounded input changes and output reads can answer the task. +- When creating spreadsheet tables, keep each table as one contiguous rectangle: title or caption, row-axis labels, column-axis labels, and data cells should be adjacent and inspectable together. Do not place row labels or key headers outside the visible table bounds or to the left of the title anchor. If you add decorative axis labels, keep the actual machine-readable row and column values inside the same table rectangle. + +## Stop rules + +- Before finalizing, inspect the requested outputs and check their location, format, shape, and content against the committed method and the task. +- If a later finding changes the method, assumptions, selections, transformations, calculations, decisions, or coverage, regenerate the affected outputs before finalizing. +- Once the requested outputs exist and the relevant checks pass, stop instead of continuing open-ended exploration. diff --git a/packages/skills/skills/data-analysis/icon.svg b/packages/skills/skills/data-analysis/icon.svg new file mode 100644 index 0000000..b8c92c8 --- /dev/null +++ b/packages/skills/skills/data-analysis/icon.svg @@ -0,0 +1,5 @@ + + + + + diff --git a/packages/skills/skills/penguin-cli/SKILL.md b/packages/skills/skills/penguin-cli/SKILL.md new file mode 100644 index 0000000..c7a1bea --- /dev/null +++ b/packages/skills/skills/penguin-cli/SKILL.md @@ -0,0 +1,69 @@ +--- +name: penguin-cli +description: Manage model API keys, default models and per-agent vault secrets with the penguin CLI. +short_description: Manage models and secrets with the penguin CLI. +short_description_zh: 用 penguin CLI 管理模型与密钥。 +version: 1 +updated: 2026-07-17T00:00:00Z +--- + +# Penguin CLI + +The `penguin` CLI manages model credentials, default models and per-agent vault secrets. Configuration goes through the CLI only — never read or hand-edit the underlying hidden files. + +## Before you start + +If the user's message only invokes this skill (e.g. "use penguin-cli skill") without a concrete request, ask the user what they want to configure. Do not run any command until the goal is clear. + +## Models + +Add or update a model (upsert by the stored model id; re-run with more options to amend an entry): + +```bash +penguin config model add --model-id [--provider ] [--api-key ] [--base-url ] \ + [--client-type ] [--context-window ] [--vision | --no-vision] \ + [--price-cache-read ] [--price-cache-write ] [--price-output ] \ + [--project-id ] [--root ] [--set-default] +``` + +- `--model-id` takes the provider's upstream model id (what the API expects). The stored id is always `/`: `--provider` picks the provider group, and when omitted it is inferred from the built-in catalog (unrecognized ids fall back to `custom`). The upstream id is persisted automatically as the entry's request id, so nothing extra is needed for it to reach the API unchanged. +- For any OpenAI chat-completion compatible endpoint use `--client-type openai --base-url `; omit `--client-type` to auto-route by model id. +- Prices are USD per million tokens (cache read / cache write / output). +- `--vision` / `--no-vision` mark whether the model accepts images; omitting both keeps the current value (default is vision-capable). +- All `penguin config model ...` and `penguin config vault ...` commands accept `--root ` to target another data root (default `PENGUIN_HOME`, then `~/.penguin/data`). + +Other model commands: + +```bash +penguin config model default --model-id --provider [--root ] # set the project default model +penguin config model vision --model-id --provider [--root ] # set the project vision model (reads images for text-only sessions) +penguin config model list [--root ] # list models; api_key is shown masked +``` + +## Vault (per-agent secrets) + +The vault holds an agent's environment-variable secrets (third-party API keys etc.); values are injected into that agent's shell subprocesses: + +```bash +penguin config vault set --key --value [--project-id ] [--agent-id ] [--root ] +penguin config vault list [--project-id ] [--agent-id ] [--root ] # values are shown masked +penguin config vault remove --key [--project-id ] [--agent-id ] [--root ] +``` + +- `--project-id` defaults to `default_project`, `--agent-id` to `default_agent`. +- Key names follow shell variable rules (letter or underscore first, then letters, digits and underscores); values are limited to 8192 characters. + +## Language + +```bash +penguin config lang # persist the CLI language via PENGUIN_LANG in your shell rc +``` + +## Running agents + +`penguin run -m "" [--model-id ] [--agent-id ] [--workspace ] [--approve ]` runs one task; `penguin chat [--resume [session_id]]` starts or resumes an interactive chat with the same options. + +## Storage + +- `/.project_config.toml` — the project's single hidden config file: model list, settings and per-model credentials (`api_key` etc. inlined in each model entry). Configuration is CLI-only — never read, print or hand-edit this file. +- `/agents//agent_state/.vault.toml` — that agent's vault entries, hidden file; same rule, manage it with `penguin config vault`. diff --git a/packages/skills/skills/penguin-cli/icon.svg b/packages/skills/skills/penguin-cli/icon.svg new file mode 100644 index 0000000..36e02fb --- /dev/null +++ b/packages/skills/skills/penguin-cli/icon.svg @@ -0,0 +1,5 @@ + + + + + diff --git a/packages/skills/skills/penguin-sdk/SKILL.md b/packages/skills/skills/penguin-sdk/SKILL.md new file mode 100644 index 0000000..7d155d1 --- /dev/null +++ b/packages/skills/skills/penguin-sdk/SKILL.md @@ -0,0 +1,88 @@ +--- +name: penguin-sdk +description: Build AI apps on the Penguin Harness SDK — self-contained projects inside the Workspace, model configuration, and the createSession/run streaming loop. +short_description: Build AI apps on the Penguin Harness SDK. +short_description_zh: 基于 Penguin Harness SDK 构建 AI 应用。 +version: 1 +updated: 2026-07-17T00:00:00Z +--- + +# Penguin Harness SDK + +`@prismshadow/penguin-core` is the TypeScript SDK this agent itself runs on. Use it to build your own AI apps: + +- An **Agent** loads its state (prompts, tools, skills) from `//agents//`. Creating an Agent whose directory is empty initializes it with defaults. +- A **Session** is one conversation of an Agent inside a **Workspace** directory. +- `session.run()` executes one task and streams every step (thinking, text, tool calls) as OmniMessages. + +To have an agent perform a task, use the `run_subagent` tool — the SDK is for building applications, not for invoking agents. + +## Before you start + +If the user's message only invokes this skill (e.g. "use penguin-sdk skill") without a concrete app to build, ask the user what they want to build. Do not start until the requirement is clear. + +## Project location + +Create the app in the current workspace directory by default (the `CWD` value from your Environment section), as a self-contained project — do not place it under `` or depend on any path outside the project folder. Point the agent data root at a directory inside the project with `createAgent({ root })`, resolved from the source file so it stays relative: + +```ts +const agent = await createAgent({ root: path.resolve(import.meta.dirname, "penguin_data") }); +``` + +With every reference relative to the project, the user can move or copy the folder anywhere and it still runs. + +## Setup + +```bash +npm install @prismshadow/penguin-core +``` + +If the package is not yet available on your npm registry (it is developed in the PenguinHarness monorepo and may not be published), develop inside a checkout of the PenguinHarness repo instead: add your app as a workspace package under `packages/` and depend on `"@prismshadow/penguin-core": "workspace:*"`, then run `pnpm install && pnpm build` at the repo root. Tell the user which route you took. + +A model must be configured for the app's data root. Two ways: + +1. The penguin CLI, pointed at the project-local data directory: + +```bash +penguin config model add --root --model-id --api-key [--base-url ] [--client-type openai] --set-default +``` + +Generally prefer the OpenAI protocol client (chat completion): `--client-type openai --base-url ` works with any OpenAI-compatible endpoint. Use exact model ids — see the agenthub-models skill for the id table. + +2. Environment variables as a fallback: without a configured credential the SDK reads the provider's env vars (e.g. `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `DEEPSEEK_API_KEY`). + +Model config lives in a single hidden file under the data root's project directory: `.project_config.toml` (model list, settings and per-model credentials such as `api_key` inlined in each model entry). It is CLI-only — never read, print or edit it; the CLI above manages it. + +## Minimal conversational app + +```ts +import path from "node:path"; +import readline from "node:readline/promises"; +import { createAgent, userText } from "@prismshadow/penguin-core"; + +const agent = await createAgent({ root: path.resolve(import.meta.dirname, "penguin_data") }); +const session = await agent.createSession({ workspaceDir: process.cwd() }); + +const rl = readline.createInterface({ input: process.stdin, output: process.stdout }); +for (;;) { + const line = await rl.question("> "); + if (!line.trim()) break; + // One run per user turn; the same Session keeps the conversation context. + for await (const msg of session.run([userText(line)], { + approve: async () => "allow", // demo only — a real app should ask its user ("deny" blocks the call) + })) { + const p = msg.payload; + if (p.type === "partial_text" && p.event_type === "delta") process.stdout.write(p.text); + } + process.stdout.write("\n"); +} +rl.close(); +``` + +Key points: + +- An Agent's behavior is edited in its `agent_state/` files (system_config.yaml, AGENTS.md, skills/), not in code. +- `createSession({ workspaceDir, modelId })` — omit `workspaceDir` for a temporary workspace, omit `modelId` for the project default model. +- `session.run(messages, { approve, signal })` is an async generator of OmniMessages; filter the payload types you care about (`partial_text` deltas carry the streamed answer). +- The `approve` callback gates every tool call — auto-allow is for demos; a production app should prompt its user before returning `"allow"`. +- Call `session.run()` again on the same Session for the next user turn — the context carries over. diff --git a/packages/skills/skills/penguin-sdk/icon.svg b/packages/skills/skills/penguin-sdk/icon.svg new file mode 100644 index 0000000..acf796b --- /dev/null +++ b/packages/skills/skills/penguin-sdk/icon.svg @@ -0,0 +1,6 @@ + + + + + + diff --git a/packages/skills/skills/software-engineering/SKILL.md b/packages/skills/skills/software-engineering/SKILL.md new file mode 100644 index 0000000..c4ee8de --- /dev/null +++ b/packages/skills/skills/software-engineering/SKILL.md @@ -0,0 +1,34 @@ +--- +name: software-engineering +description: Complete software-engineering tasks — investigate and review code, implement bug fixes, features and refactors with minimal scope, validate changes, and report verified outcomes. +short_description: Complete software-engineering tasks. +short_description_zh: 完成软件工程任务。 +version: 1 +updated: 2026-07-18T00:00:00Z +--- + +# Software Engineering + +This skill guides PenguinHarness through general software-engineering work, including code investigation, reviews, bug fixes, features, refactors, and verified handoff. + +## Before you start + +If the user's message only invokes this skill without a concrete software-engineering task, ask what they want investigated, reviewed, fixed, or implemented. Do not start until the task is clear. + +## Workflow + +- Match the work to the user's intent. For explanation, review, or planning requests, inspect and report without editing unless a change is also requested. For implementation tasks, carry the requested change through verification and handoff. +- Use the current working directory as the target project and work in place unless the user directs otherwise. +- Preserve worktree changes you did not make. Never revert unrelated user or collaborator work, and do not use destructive Git commands unless explicitly requested. +- Act autonomously and use tools to understand the real code. Resolve minor ambiguity from existing behavior, tests, and repository conventions; ask only when different choices would materially change the requested behavior or scope. +- Before making changes, read the applicable repository instructions (such as `AGENTS.md` or `CLAUDE.md`) and identify the repository-provided build, test, lint, and formatting commands. +- Inspect the relevant implementation, callers, tests, and interfaces before making changes. +- For bug reports, try to reproduce the issue or identify a failing test before editing when feasible. Use the reproduced behavior or failing test to verify the fix afterward. +- Make the smallest coherent change that fully satisfies the request. Follow existing patterns and dependencies, preserve unrelated behavior, and update coupled tests, schemas, configuration, or generated artifacts only when the repository requires it. Never weaken a test to justify the implementation. +- For implementation tasks, do not stop after analysis. Inspect failures, revise the approach, and continue until the change is verified or a concrete blocker remains. Do not repeat a failed action without changing the approach. +- Validate code changes proportionally with repository-provided commands: run the most focused relevant check first and broaden only when useful. Preserve exit status, inspect relevant failure output, and never claim that an unobserved check passed. +- Before finishing a code change, review `git diff` and repository status, remove temporary artifacts, and keep the change limited to the task. Do not commit unless the user or applicable repository instructions require it. + +## Handoff + +- Keep the final response concise: summarize the outcome or change, list checks actually run and their outcomes, and state any remaining limitation. diff --git a/packages/skills/skills/software-engineering/icon.svg b/packages/skills/skills/software-engineering/icon.svg new file mode 100644 index 0000000..e4a3759 --- /dev/null +++ b/packages/skills/skills/software-engineering/icon.svg @@ -0,0 +1,5 @@ + + + + + diff --git a/packages/skills/skills/web-design/SKILL.md b/packages/skills/skills/web-design/SKILL.md new file mode 100644 index 0000000..db57a28 --- /dev/null +++ b/packages/skills/skills/web-design/SKILL.md @@ -0,0 +1,40 @@ +--- +name: web-design +description: Default visual language for generated web pages — minimal black-white-gray, square corners, hierarchy from weight and spacing; colors, radii and gradients only on explicit request. +short_description: Minimal monochrome defaults for generated web pages. +short_description_zh: 生成网页的极简黑白默认视觉规范。 +version: 1 +updated: 2026-07-17T00:00:00Z +--- + +# Web Design + +Default visual rules for every web page or frontend interface you generate. Apply them to any HTML/CSS you produce unless the user explicitly asks otherwise. + +## Before you start + +If the user's message only invokes this skill (e.g. "use web-design skill") without a concrete page or interface to build, ask the user what they want to build. Do not start until the requirement is clear. + +## Core rules + +- Monochrome only: black, white and grays. No accent colors, no gradients, no decorative shadows. +- Square corners everywhere: `border-radius: 0` on buttons, cards, inputs, images and modals. +- Hierarchy comes from font weight, font size, spacing and thin light-gray borders — never from colored blocks or backgrounds. +- Generous whitespace: prefer more spacing over more dividers; let sections breathe. +- Introduce colors, rounded corners or gradients **only when the user explicitly asks for them**, and only where asked — the rest of the page stays monochrome and square. + +## Tokens + +Base every stylesheet on a small monochrome token set: + +```css +:root { + --fg: #111111; /* primary text */ + --fg-muted: #666666; /* secondary text */ + --bg: #ffffff; /* page background */ + --bg-subtle: #f5f5f5; /* raised surfaces */ + --border: #e2e2e2; /* hairline borders (1px) */ +} +``` + +Tailwind equivalent: stick to `text-neutral-900` / `text-neutral-500` / `bg-white` / `bg-neutral-100` / `border-neutral-200` / `rounded-none`; do not use color utilities (`blue-*`, `emerald-*`, ...), `rounded-*` variants other than `rounded-none`, or `bg-gradient-*`. diff --git a/packages/skills/skills/web-design/icon.svg b/packages/skills/skills/web-design/icon.svg new file mode 100644 index 0000000..40140d4 --- /dev/null +++ b/packages/skills/skills/web-design/icon.svg @@ -0,0 +1,9 @@ + + + + + + + + + diff --git a/packages/skills/src/index.ts b/packages/skills/src/index.ts new file mode 100644 index 0000000..4ab175f --- /dev/null +++ b/packages/skills/src/index.ts @@ -0,0 +1,212 @@ +/** + * PenguinHarness Skill library: built-in SKILL.md docs and skill group manifest. + * + * The runtime source of truth for library content is the package's `skills//SKILL.md` + * files: frontmatter is parsed on read (same rules as installed Skills), so editing a file takes + * effect immediately with no caching (files are small, calls are infrequent). Only the skill group + * manifest (id, title, and member names) is hardcoded in code; install / uninstall / scan still live + * in core's state layer. + * + * Docs: packages/docs/content/skills.{zh,en}.md (site path /docs/skills) documents the Skill + * format and the built-in library. + */ +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +/** Skill's frontmatter metadata (four fields: name / description / version / updated; description itself is English-only, may also carry description_zh and short_description(_zh)). */ +export interface SkillMetadata { + /** Skill name (matches its containing directory name). */ + name: string; + /** One-line description; injected into the model prompt via `{{SKILL_METADATA}}`. */ + description: string; + /** UI short description (frontmatter `short_description`, optional): preferred in compact spots like cards, falls back to the full description if missing; not injected into the prompt. */ + shortDescription?: string; + /** Chinese short description (frontmatter `short_description_zh`, optional). */ + shortDescriptionZh?: string; + /** Version number (natural number); falls back to 1 on parse failure. */ + version: number; + /** Update date (YYYY-MM-DD); defaults to "". */ + updated: string; +} + +/** A Skill in the library: metadata + full SKILL.md content (including frontmatter, written as-is on install). */ +export interface LibrarySkill extends SkillMetadata { + content: string; + /** Optional raw `icon.svg` content in the directory (custom icon, the file is the sole source, copied alongside SKILL.md on install); absent means none (frontend falls back to the default book icon). */ + icon?: string; +} + +/** Skill group manifest entry: group id, title (optionally with a Chinese title, displayed per UI language), and member Skill names. */ +export interface SkillGroupInfo { + id: string; + title: string; + /** Chinese group title (optional, displayed per UI language). */ + titleZh?: string; + /** Member Skill names (i.e., directory names under `skills/`). */ + skills: string[]; +} + +/** Grouping result: group metadata + member Skills read from library files. */ +export interface ResolvedSkillGroup extends Omit { + skills: LibrarySkill[]; +} + +/** + * Parses the frontmatter at the start of SKILL.md: only recognizes `key: value` lines inside the + * first `---` block (split on the first colon, value trimmed, values may themselves contain colons); + * all fields are scalars, no YAML dependency needed. + * Error tolerance: returns null if the `---` block or name is missing; version falls back to 1 if + * it isn't a natural number; updated defaults to "". + */ +export function parseSkillFrontmatter(content: string): SkillMetadata | null { + // Strip a possible UTF-8 BOM (may be introduced by editors when manually editing an installed SKILL.md); CRLF is handled by \r?\n. + const match = /^---\r?\n([\s\S]*?)\r?\n---/.exec(content.replace(/^\uFEFF/, "")); + if (!match) return null; + const fields: Record = {}; + for (const line of match[1]!.split(/\r?\n/)) { + const idx = line.indexOf(":"); + if (idx <= 0) continue; + const key = line.slice(0, idx).trim(); + if (key) fields[key] = line.slice(idx + 1).trim(); + } + const name = fields["name"]; + if (!name) return null; + const version = Number.parseInt(fields["version"] ?? "", 10); + const shortDescription = fields["short_description"]; + const shortDescriptionZh = fields["short_description_zh"]; + return { + name, + description: fields["description"] ?? "", + // short_description(_zh) is optional: omitted when absent (undefined keys aren't set). + ...(shortDescription !== undefined ? { shortDescription } : {}), + ...(shortDescriptionZh !== undefined ? { shortDescriptionZh } : {}), + version: Number.isInteger(version) && version >= 1 ? version : 1, + updated: fields["updated"] ?? "", + }; +} + +/** Root directory of library files: the package's `skills/` (both dist/ and src/ sit one level below the package root, so one level up reaches it). */ +const SKILLS_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "skills"); + +/** Character rule for Skill names (directory names): prevents path traversal. */ +const SKILL_NAME_PATTERN = /^[A-Za-z0-9_-]+$/; + +/** + * Reads a single library directory to construct a LibrarySkill; returns undefined if SKILL.md + * doesn't exist. name is taken from the directory name (overriding frontmatter); falls back to + * empty metadata if frontmatter parsing fails. + * The optional icon.svg in the directory is read alongside it (as a raw string); the icon field + * is omitted if missing. + */ +function readSkillDir(name: string): LibrarySkill | undefined { + const dir = path.join(SKILLS_ROOT, name); + let content: string; + try { + content = fs.readFileSync(path.join(dir, "SKILL.md"), "utf8"); + } catch { + return undefined; + } + let icon: string | undefined; + try { + icon = fs.readFileSync(path.join(dir, "icon.svg"), "utf8"); + } catch { + // icon.svg is optional: no custom icon if missing. + } + const meta = parseSkillFrontmatter(content) ?? { + name, + description: "", + version: 1, + updated: "", + }; + return { ...meta, name, content, ...(icon !== undefined ? { icon } : {}) }; +} + +/** Reads all Skills in the library (one per subdirectory under `skills/`), sorted by name. */ +export function loadLibrarySkills(): LibrarySkill[] { + const skills: LibrarySkill[] = []; + for (const entry of fs.readdirSync(SKILLS_ROOT, { withFileTypes: true })) { + if (!entry.isDirectory()) continue; + const skill = readSkillDir(entry.name); + if (skill) skills.push(skill); + } + return skills.sort((a, b) => a.name.localeCompare(b.name)); +} + +/** + * Skill group manifest; members are library directory names. + * Docs: /docs/skills § "Built-in library". + */ +export const SKILL_GROUPS: SkillGroupInfo[] = [ + { + id: "agent-development", + title: "Agent Development", + titleZh: "Agent 开发", + skills: ["agent-creation", "benchmark-design", "agent-evaluation", "agent-optimization"], + }, + { + id: "data-analysis", + title: "Data Analysis", + titleZh: "数据分析", + skills: ["data-analysis"], + }, + { + id: "penguin-development", + title: "Penguin Development", + titleZh: "Penguin 开发", + skills: ["penguin-sdk", "penguin-cli", "agenthub-models"], + }, + { + id: "web-development", + title: "Web Development", + titleZh: "网页开发", + skills: ["web-design"], + }, + { + id: "software-engineering", + title: "Software Engineering", + titleZh: "软件工程", + skills: ["software-engineering"], + }, +]; + +/** + * Groups library Skills according to SKILL_GROUPS (a member name missing from `all` is skipped); + * Skills not listed in any group are appended to an Other group (only appears if non-empty). A + * pure function, the testable core of loadSkillGroups. + */ +export function groupSkills(all: LibrarySkill[]): ResolvedSkillGroup[] { + const byName = new Map(all.map((skill) => [skill.name, skill])); + const grouped = new Set(); + const groups: ResolvedSkillGroup[] = SKILL_GROUPS.map((group) => { + const members: LibrarySkill[] = []; + for (const name of group.skills) { + const skill = byName.get(name); + if (!skill) continue; + members.push(skill); + grouped.add(name); + } + return { ...group, skills: members }; + }); + const others = all.filter((skill) => !grouped.has(skill.name)); + if (others.length > 0) { + groups.push({ + id: "other", + title: "Other", + titleZh: "其他", + skills: others, + }); + } + return groups; +} + +/** Reads library files and groups them: SKILL_GROUPS order comes first, ungrouped Skills are appended to an Other group (only appears if non-empty). */ +export function loadSkillGroups(): ResolvedSkillGroup[] { + return groupSkills(loadLibrarySkills()); +} + +/** Reads a single library Skill by name; returns undefined if the name contains illegal characters (path traversal guard) or doesn't exist. */ +export function librarySkill(name: string): LibrarySkill | undefined { + if (!SKILL_NAME_PATTERN.test(name)) return undefined; + return readSkillDir(name); +} diff --git a/packages/skills/test/skills.test.ts b/packages/skills/test/skills.test.ts new file mode 100644 index 0000000..582322c --- /dev/null +++ b/packages/skills/test/skills.test.ts @@ -0,0 +1,243 @@ +/** + * Tests for the Skill library file source of truth and its parser: loadLibrarySkills reading + * files into a manifest, loadSkillGroups grouping, groupSkills' Other group and missing-member + * tolerance, librarySkill's traversal-name rejection, doc conventions (`## Before you start` is + * mandatory), and parseSkillFrontmatter's error tolerance. + */ +import fs from "node:fs/promises"; +import path from "node:path"; +import { describe, expect, it } from "vitest"; +import { + SKILL_GROUPS, + groupSkills, + librarySkill, + loadLibrarySkills, + loadSkillGroups, + parseSkillFrontmatter, + type LibrarySkill, +} from "../src/index.js"; + +const skillsRoot = path.resolve(import.meta.dirname, "../skills"); + +/** Minimal LibrarySkill for groupSkills unit tests. */ +const fakeSkill = (name: string): LibrarySkill => ({ + name, + description: `Do ${name}.`, + version: 1, + updated: "2026-07-17T00:00:00Z", + content: `---\nname: ${name}\n---\nBody`, +}); + +describe("loadLibrarySkills", () => { + it("按 name 排序读出技能,metadata 齐全(含中文描述与短描述)", async () => { + const skills = loadLibrarySkills(); + const names = skills.map((skill) => skill.name); + expect(names).toEqual([...names].sort()); + for (const skill of skills) { + expect(skill.description, skill.name).toBeTruthy(); + // Short description (UI display): both languages present, and clearly shorter than the full description. + expect(skill.shortDescription, skill.name).toBeTruthy(); + expect(skill.shortDescriptionZh, skill.name).toBeTruthy(); + expect(skill.shortDescription!.length, skill.name).toBeLessThan(skill.description.length); + expect(skill.shortDescriptionZh!.length, skill.name).toBeLessThan(skill.description.length); + // Pre-release, version is always 1. + expect(skill.version).toBe(1); + expect(skill.updated).toMatch(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?Z$/); + // content is the full SKILL.md text including frontmatter (written as-is on install). + expect(skill.content.startsWith("---\n")).toBe(true); + } + }); + + it("每个技能都带定制 icon.svg(原文读入,站内线稿风,无脚本)", async () => { + for (const skill of loadLibrarySkills()) { + const raw = await fs.readFile(path.join(skillsRoot, skill.name, "icon.svg"), "utf8"); + // The icon field is the raw icon.svg content in the directory (the file is the sole source). + expect(skill.icon, skill.name).toBe(raw); + expect(skill.icon, skill.name).toContain('viewBox="0 0 24 24"'); + expect(skill.icon, skill.name).toContain('stroke="currentColor"'); + expect(skill.icon, skill.name).toContain('fill="none"'); + // Security baseline: no scripts or event attributes (frontend also sanitizes before inline rendering). + expect(skill.icon, skill.name).not.toContain(" { + const dirs = (await fs.readdir(skillsRoot, { withFileTypes: true })) + .filter((entry) => entry.isDirectory()) + .map((entry) => entry.name) + .sort(); + const skills = loadLibrarySkills(); + expect(skills.map((s) => s.name)).toEqual(dirs); + for (const skill of skills) { + const raw = await fs.readFile(path.join(skillsRoot, skill.name, "SKILL.md"), "utf8"); + expect(skill.content).toBe(raw); + // The library file's own frontmatter name should match its directory name (content quality constraint). + expect(skill.content).toContain(`name: ${skill.name}`); + } + }); + + it("每个技能正文都有 `## Before you start` 一节(无具体需求时先反问)", () => { + for (const skill of loadLibrarySkills()) { + expect(skill.content, skill.name).toContain("## Before you start"); + } + }); +}); + +describe("loadSkillGroups / groupSkills", () => { + it("按 Skill 组清单解组,成员齐全且带中文组名,无 Other 组", () => { + const groups = loadSkillGroups(); + expect(groups.map((g) => g.id)).toEqual([ + "agent-development", + "data-analysis", + "penguin-development", + "web-development", + "software-engineering", + ]); + expect(groups[0]!.skills.map((s) => s.name)).toEqual([ + "agent-creation", + "benchmark-design", + "agent-evaluation", + "agent-optimization", + ]); + expect(groups[1]!.skills.map((s) => s.name)).toEqual(["data-analysis"]); + expect(groups[1]!.title).toBe("Data Analysis"); + expect(groups[1]!.titleZh).toBe("数据分析"); + expect(groups[2]!.skills.map((s) => s.name)).toEqual([ + "penguin-sdk", + "penguin-cli", + "agenthub-models", + ]); + expect(groups[3]!.skills.map((s) => s.name)).toEqual(["web-design"]); + expect(groups[3]!.title).toBe("Web Development"); + expect(groups[3]!.titleZh).toBe("网页开发"); + expect(groups[4]!.skills.map((s) => s.name)).toEqual(["software-engineering"]); + expect(groups[4]!.title).toBe("Software Engineering"); + expect(groups[4]!.titleZh).toBe("软件工程"); + for (const group of groups) { + expect(group.title).toBeTruthy(); + expect(group.titleZh).toBeTruthy(); + // Groups no longer carry a description (group header is just title + skill count). + expect("description" in group).toBe(false); + } + }); + + it("groupSkills:未列入任何组的技能追加 Other 组(含中英文组名)", () => { + const stray = fakeSkill("stray-skill"); + const groups = groupSkills([fakeSkill("agent-creation"), stray]); + expect(groups.map((g) => g.id)).toEqual([ + "agent-development", + "data-analysis", + "penguin-development", + "web-development", + "software-engineering", + "other", + ]); + const other = groups[5]!; + expect(other.title).toBe("Other"); + expect(other.titleZh).toBe("其他"); + expect(other.skills).toEqual([stray]); + }); + + it("groupSkills:成员名缺失则跳过;全部入组时不出现 Other 组", () => { + const groups = groupSkills([fakeSkill("penguin-cli")]); + expect(groups.map((g) => g.id)).toEqual([ + "agent-development", + "data-analysis", + "penguin-development", + "web-development", + "software-engineering", + ]); + expect(groups[0]!.skills).toEqual([]); + expect(groups[1]!.skills).toEqual([]); + expect(groups[2]!.skills.map((s) => s.name)).toEqual(["penguin-cli"]); + expect(groups[3]!.skills).toEqual([]); + expect(groups[4]!.skills).toEqual([]); + }); + + it("SKILL_GROUPS 清单硬编码为成员名(库文件之外唯一的组信息真源)", () => { + expect(SKILL_GROUPS.map((g) => ({ id: g.id, skills: g.skills }))).toEqual([ + { + id: "agent-development", + skills: ["agent-creation", "benchmark-design", "agent-evaluation", "agent-optimization"], + }, + { id: "data-analysis", skills: ["data-analysis"] }, + { id: "penguin-development", skills: ["penguin-sdk", "penguin-cli", "agenthub-models"] }, + { id: "web-development", skills: ["web-design"] }, + { id: "software-engineering", skills: ["software-engineering"] }, + ]); + }); +}); + +describe("librarySkill", () => { + it("按名称读单个技能,未知名称返回 undefined", () => { + expect(librarySkill("penguin-sdk")?.name).toBe("penguin-sdk"); + expect(librarySkill("no-such-skill")).toBeUndefined(); + }); + + it("非法字符名一律拒绝(防路径穿越),不触达文件系统", () => { + for (const name of ["../penguin-sdk", "..", "penguin-sdk/SKILL.md", "a/../b", ".", ""]) { + expect(librarySkill(name), name).toBeUndefined(); + } + }); +}); + +describe("parseSkillFrontmatter", () => { + it("解析 name/description/version/updated,值允许含冒号", () => { + const meta = parseSkillFrontmatter( + "---\nname: demo\ndescription: How to use x: y and z\nversion: 3\nupdated: 2026-07-16\n---\n\nBody", + ); + expect(meta).toEqual({ + name: "demo", + description: "How to use x: y and z", + version: 3, + updated: "2026-07-16", + }); + }); + + it("short_description_zh 可选:有则解析,缺省不带该字段", () => { + const withZh = parseSkillFrontmatter( + "---\nname: demo\ndescription: Do x\nshort_description_zh: 做 x\n---\nBody", + ); + expect(withZh?.shortDescriptionZh).toBe("做 x"); + const withoutZh = parseSkillFrontmatter("---\nname: demo\ndescription: Do x\n---\nBody"); + expect(withoutZh).not.toBeNull(); + expect(withoutZh && "shortDescriptionZh" in withoutZh).toBe(false); + }); + + it("short_description(_zh) 可选:有则解析为 shortDescription(Zh),缺省不带该字段", () => { + const withShort = parseSkillFrontmatter( + "---\nname: demo\ndescription: Do x in detail\nshort_description: Do x\nshort_description_zh: 做 x\n---\nBody", + ); + expect(withShort?.shortDescription).toBe("Do x"); + expect(withShort?.shortDescriptionZh).toBe("做 x"); + const without = parseSkillFrontmatter("---\nname: demo\ndescription: Do x\n---\nBody"); + expect(without && "shortDescription" in without).toBe(false); + expect(without && "shortDescriptionZh" in without).toBe(false); + }); + + it("UTF-8 BOM 与 CRLF 换行照常解析(手改文件的编辑器可能引入)", () => { + const bom = parseSkillFrontmatter("\uFEFF---\nname: demo\ndescription: Do x\n---\nBody"); + expect(bom?.name).toBe("demo"); + const crlf = parseSkillFrontmatter("---\r\nname: demo\r\ndescription: Do x\r\n---\r\nBody"); + expect(crlf?.description).toBe("Do x"); + }); + + it("缺 --- 块或缺 name 返回 null", () => { + expect(parseSkillFrontmatter("# No frontmatter")).toBeNull(); + expect(parseSkillFrontmatter("---\ndescription: only desc\n---\nBody")).toBeNull(); + // A block that isn't at the start doesn't count as frontmatter either. + expect(parseSkillFrontmatter("Body\n---\nname: x\n---")).toBeNull(); + }); + + it("version 非自然数回退 1,updated 缺省空串", () => { + expect(parseSkillFrontmatter("---\nname: a\nversion: zero\n---")?.version).toBe(1); + expect(parseSkillFrontmatter("---\nname: a\nversion: 0\n---")?.version).toBe(1); + expect(parseSkillFrontmatter("---\nname: a\n---")).toEqual({ + name: "a", + description: "", + version: 1, + updated: "", + }); + }); +}); diff --git a/packages/skills/tsconfig.json b/packages/skills/tsconfig.json new file mode 100644 index 0000000..8cd1715 --- /dev/null +++ b/packages/skills/tsconfig.json @@ -0,0 +1,7 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { + "rootDir": "." + }, + "include": ["src", "test"] +} diff --git a/packages/skills/tsup.config.ts b/packages/skills/tsup.config.ts new file mode 100644 index 0000000..75bf903 --- /dev/null +++ b/packages/skills/tsup.config.ts @@ -0,0 +1,10 @@ +import { defineConfig } from "tsup"; + +export default defineConfig({ + entry: ["src/index.ts"], + format: ["esm"], + target: "node20", + dts: true, + clean: true, + sourcemap: true, +}); diff --git a/packages/web/.gitignore b/packages/web/.gitignore new file mode 100644 index 0000000..11c8230 --- /dev/null +++ b/packages/web/.gitignore @@ -0,0 +1,3 @@ +# Playwright E2E artifacts +test-results/ +playwright-report/ diff --git a/packages/web/README.md b/packages/web/README.md new file mode 100644 index 0000000..9d0b092 --- /dev/null +++ b/packages/web/README.md @@ -0,0 +1,43 @@ +# @prismshadow/penguin-web + +The PenguinHarness Web App — a React 19 + Vite + Tailwind CSS 4 SPA that renders the OmniMessage stream (same protocol and statistics as the CLI) and manages Agents, Skills, Models, usage and Traces. Feature tour: [Web App Guide](https://prism-shadow.github.io/penguin-harness/docs/web-app). + +## Layout + +``` +src/ +├── main.tsx / app.tsx / router.tsx / styles.css +├── api/ # fetch wrapper, typed endpoint functions, EventSource (SSE) wrapper +├── state/ # auth / project / sessions / theme / locale contexts +├── lib/ +│ ├── omni/ # OmniMessage stream → view-model reducer + connect-first/dedup controller +│ └── … # formatting, i18n dictionaries (zh/en), attachments, helpers +├── components/ # ui primitives (modal, drawer, select, …) + app layout +└── features/ # chat / agents / skills / models / usage / traces / benchmark / admin +``` + +DTO types are imported type-only from `@prismshadow/penguin-server/api`; no server code enters the bundle. Rendering rules for streaming partials (start/delta/stop aggregation, complete-message replacement, origin-chain nesting into subagent cards) live in `lib/omni/stream-model.ts`, which is fully unit-tested. + +## Development + +Prereqs: Node >= 24, pnpm; run `pnpm install` at the repo root first (core must be built — the root `dev:*` scripts handle that). + +```bash +pnpm dev:server # backend at 127.0.0.1:7364 +pnpm dev:web # Vite dev server at 127.0.0.1:7365; /api proxied (SSE passes through) +``` + +The proxy target defaults to `http://127.0.0.1:7364` (`PENGUIN_API_PROXY` overrides). Auth is a same-origin HttpOnly cookie, so the proxy keeps everything same-origin. + +```bash +pnpm --filter @prismshadow/penguin-web typecheck +pnpm --filter @prismshadow/penguin-web test # vitest (pure logic) +pnpm --filter @prismshadow/penguin-web test:e2e # Playwright against a mock LLM +pnpm --filter @prismshadow/penguin-web build # vite build → dist/ +``` + +## Production + +No separate static server needed: `@prismshadow/penguin-server` auto-hosts `packages/web/dist` (or `PENGUIN_WEB_DIST`) with an SPA fallback — build the web app, start the server, done. The published npm packages bundle the built front end. + +Part of [PenguinHarness](https://github.com/Prism-Shadow/penguin-harness) · Apache-2.0 diff --git a/packages/web/e2e/README.md b/packages/web/e2e/README.md new file mode 100644 index 0000000..635eb1c --- /dev/null +++ b/packages/web/e2e/README.md @@ -0,0 +1,13 @@ +# Web E2E(Playwright) + +浏览器端到端:对话(thinking + 工具审批 + 工具执行 + 二轮答复)、图标化统计(成本折算 / +复制回复)、轨迹观测(分 Task 时间线 + 图例 + 悬停联动高亮)、Workspace 文件预览(HTML +sandbox 渲染、路径默认隐藏)。LLM 由 `mock-llm.mjs`(mock Anthropic Messages SSE)驱动, +不联网。 + +```sh +pnpm --filter @prismshadow/penguin-web test:e2e # 构建 + 起服务 + 跑用例 +SKIP_BUILD=1 pnpm --filter @prismshadow/penguin-web test:e2e # 跳过构建 +``` + +首次需 `npx playwright install chromium`。 diff --git a/packages/web/e2e/auth.mjs b/packages/web/e2e/auth.mjs new file mode 100644 index 0000000..6cc4850 --- /dev/null +++ b/packages/web/e2e/auth.mjs @@ -0,0 +1,38 @@ +/** + * e2e auth helper: with signup disabled, test users are always provisioned via + * the built-in admin account, then logged in. The server seeds an admin + * (admin / admin123) on startup; a single e2e run shares one data root, and + * provisioning is idempotent (reuses the user if it already exists) so a + * single spec can be rerun on its own. + */ +import { request } from "@playwright/test"; + +const BASE = process.env.BASE_URL; +export const ADMIN_ID = "admin"; +export const ADMIN_PASSWORD = "admin123"; + +/** Log in: the cookie lands in the given request context (page.request is the browser context); returns user. */ +export async function login(ctx, userId, password) { + const res = await ctx.post(`${BASE}/api/auth/login`, { data: { userId, password } }); + if (!res.ok()) { + throw new Error(`login ${userId} failed: ${res.status()} ${await res.text()}`); + } + return (await res.json()).user; +} + +/** Admin creates the user (409 is treated as already-exists, idempotent). */ +export async function provisionUser(userId, password) { + const adminCtx = await request.newContext(); + await login(adminCtx, ADMIN_ID, ADMIN_PASSWORD); + const created = await adminCtx.post(`${BASE}/api/admin/users`, { data: { userId, password } }); + if (!created.ok() && created.status() !== 409) { + throw new Error(`create user ${userId} failed: ${created.status()} ${await created.text()}`); + } + await adminCtx.dispose(); +} + +/** Provision the user and log ctx in as them; returns user. */ +export async function provisionAndLogin(ctx, userId, password) { + await provisionUser(userId, password); + return login(ctx, userId, password); +} diff --git a/packages/web/e2e/chat.spec.mjs b/packages/web/e2e/chat.spec.mjs new file mode 100644 index 0000000..8721a55 --- /dev/null +++ b/packages/web/e2e/chat.spec.mjs @@ -0,0 +1,323 @@ +import { test, expect } from "@playwright/test"; +import { provisionAndLogin } from "./auth.mjs"; + +const BASE = process.env.BASE_URL; +const MOCK = process.env.MOCK_URL; +const U = "e2euser"; +const P = "password123"; + +test("chat + tool approval + stats/cost/copy + traces + files", async ({ page }) => { + // --- seed via API (cookies land in the browser context) --- + await provisionAndLogin(page.request, U, P); + + const projects = await (await page.request.get(`${BASE}/api/projects`)).json(); + const projectId = projects.projects[0].projectId; + + const put = await page.request.put(`${BASE}/api/projects/${projectId}/models`, { + data: { + defaultModel: { provider: "custom", modelId: "claude-4-8" }, + models: [ + { + provider: "custom", + modelId: "claude-4-8", + apiKey: "sk-mock", + baseUrl: MOCK, + contextWindow: 200000, + pricing: { cacheRead: 1, cacheWrite: 5, output: 10 }, + }, + ], + }, + }); + expect(put.ok(), "put models").toBeTruthy(); + + const agents = await (await page.request.get(`${BASE}/api/projects/${projectId}/agents`)).json(); + // The project ships with exactly one builtin agent: default_agent. + const agentIds = agents.agents.map((a) => a.agentId); + expect(agentIds).toEqual(["default_agent"]); + const agentId = "default_agent"; + + // Approval defaults to allow-all; this test verifies the manual approval flow, so specify always-ask explicitly. + const sess = await ( + await page.request.post(`${BASE}/api/projects/${projectId}/agents/${agentId}/sessions`, { + data: { provider: "custom", modelId: "claude-4-8", approvalMode: "always-ask" }, + }) + ).json(); + const sessionId = sess.session.sessionId; + + // --- chat --- + await page.goto(`${BASE}/chat/${sessionId}`); + const ta = page.getByPlaceholder(/输入消息/); + await ta.waitFor(); + + // Approval-mode is a custom dropdown (button), NOT a native setIdInput(e.target.value)} + autoFocus + /> + + + {S.project.idPrefixHint} + + + ) : ( + setIdInput(e.target.value)} + hint={S.project.idHint} + autoFocus + /> + )} + setName(e.target.value)} + /> + {error &&

    {error}

    } + + + ); +} + +/** Project settings dialog: member management (owner) and deletion (owner); members see a read-only member list. */ +export function ProjectSettingsDialog({ open, onClose }: { open: boolean; onClose: () => void }) { + const { user } = useAuth(); + const { currentProject, setCurrentProjectId, projects, reloadProjects } = useProject(); + const [members, setMembers] = useState(null); + const [newMemberId, setNewMemberId] = useState(""); + const [error, setError] = useState(null); + const [confirmDelete, setConfirmDelete] = useState(false); + + const projectId = currentProject?.projectId; + const isOwner = currentProject?.role === "owner"; + + useEffect(() => { + if (!open || !projectId) return; + setMembers(null); + setError(null); + setConfirmDelete(false); + api + .listMembers(projectId) + .then((res) => setMembers(res.members)) + .catch((e: unknown) => setError(e instanceof ApiError ? e.message : S.common.unknownError)); + }, [open, projectId]); + + if (!currentProject || !projectId) return null; + + const addMember = async () => { + if (!newMemberId.trim()) return; + setError(null); + try { + await api.addMember(projectId, { userId: newMemberId.trim() }); + setNewMemberId(""); + const res = await api.listMembers(projectId); + setMembers(res.members); + } catch (e) { + setError(e instanceof ApiError ? e.message : S.common.unknownError); + } + }; + + const doRemove = async (memberId: string) => { + setError(null); + try { + await api.removeMember(projectId, memberId); + const res = await api.listMembers(projectId); + setMembers(res.members); + } catch (e) { + setError(e instanceof ApiError ? e.message : S.common.unknownError); + } + }; + + const doDelete = async () => { + setError(null); + try { + await api.deleteProject(projectId); + onClose(); + const next = projects.find((p) => p.projectId !== projectId); + await reloadProjects(); + if (next) setCurrentProjectId(next.projectId); + } catch (e) { + setError(e instanceof ApiError ? e.message : S.common.unknownError); + } + }; + + return ( + +
    +
    +

    {S.project.switcher}

    +

    + {projectDisplayName(currentProject)}{" "} + {projectId} +

    +
    + +
    +

    {S.project.members}

    + {members === null ? ( +

    {S.common.loading}

    + ) : ( + // Member permission table: username / role / actions; cells never wrap. + // Last row (owner only) = add member: small username input + add button (new members are always the member role). +
    + + + + + + + + + + {members.map((m) => ( + + + + + + ))} + {isOwner && ( + + + + + + )} + +
    + {S.project.memberUsername} + + {S.project.memberRole} + + {S.project.memberActions} +
    {m.userId} + {m.role} + + {isOwner && m.role !== "owner" && m.userId !== user?.userId && ( + + )} +
    + setNewMemberId(e.target.value)} + onKeyDown={(e) => { + if (e.key === "Enter") void addMember(); + }} + /> + + member + + +
    +
    + )} +
    + + {isOwner && ( +
    + {projectId === "default_project" ? ( +

    {S.project.deleteDefaultForbidden}

    + ) : projects.length <= 1 ? ( + // Last accessible Project: deleting it would leave the account with no Project to select + // (the page would get stuck on the skeleton screen), so the frontend hides the entry point outright, matching the server's 409 rejection. +

    {S.project.deleteLastForbidden}

    + ) : confirmDelete ? ( +
    +

    {S.project.deleteConfirm}

    +
    + + +
    +
    + ) : ( + + )} +
    + )} + + {error &&

    {error}

    } +
    +
    + ); +} diff --git a/packages/web/src/components/layout/sidebar.tsx b/packages/web/src/components/layout/sidebar.tsx new file mode 100644 index 0000000..92801fc --- /dev/null +++ b/packages/web/src/components/layout/sidebar.tsx @@ -0,0 +1,818 @@ +/** + * Single-column sidebar, top to bottom: + * Project switcher -> new chat (default_agent draft) + fixed nav (Agents / models / cost center / + * Trace) -> Session area grouped by Agent (group header = Agent name + new chat + Agent settings; + * shows all Agents, including empty groups) -> bottom user config (theme / language / logout). + * Desktop keeps it pinned as the left column; mobile puts the whole thing in a drawer. + * New chats always enter draft state (/chat/new, route state specifies the Agent): Model / + * Workspace / approval mode are all chosen on the draft input card, so there's no longer a + * separate "quick / advanced" pair of new-chat dialogs. + * Color scheme is white/gray-based: active state uses a solid gray fill, running status uses a small color dot, no large blocks of color. + */ +import { useState } from "react"; +import type { ReactNode } from "react"; +import { NavLink, useMatch, useNavigate } from "react-router"; +import type { SessionInfo } from "@prismshadow/penguin-server/api"; +import * as api from "../../api/endpoints"; +import { ApiError } from "../../api/client"; +import { S } from "../../lib/strings"; +import { useAuth } from "../../state/auth"; +import { useLocale } from "../../state/locale"; +import type { LangPref } from "../../state/locale"; +import { ACCENT_SWATCHES, useTheme } from "../../state/theme"; +import type { Accent, Currency, FontScale, ThemeMode } from "../../state/theme"; +import { agentDisplayName, projectDisplayName, useProject } from "../../state/project"; +import { useSessions } from "../../state/sessions"; +import { Dropdown } from "../ui/dropdown"; +import { AgentAvatar } from "../ui/agent-avatar"; +import { Chevron } from "../ui/chevron"; +import { Truncated } from "../ui/truncated"; +import { Badge } from "../ui/badge"; +import { Modal } from "../ui/modal"; +import { Button } from "../ui/button"; +import { Input } from "../ui/input"; +import { Segmented } from "../ui/segmented"; +import { SkeletonList } from "../ui/skeleton"; +import { DRAFT_SESSION_ID } from "../../features/chat/chat-page"; +import { clearDraft, sessionDraftKey } from "../../features/chat/draft-cache"; +import { CreateProjectDialog, ProjectSettingsDialog } from "./project-dialogs"; +import { ChangePasswordDialog } from "../account/change-password-dialog"; + +function Icon({ d, size = 16 }: { d: string; size?: number }) { + return ( + + + + ); +} + +/** Dropdown caret (used by the Project switcher; distinct from the collapse-indicator Chevron). */ +function DropdownCaret() { + return ( + + + + ); +} + +const NAV_ICONS = { + agents: "M12 3v3m-6 4a6 6 0 0 1 12 0v5a3 3 0 0 1-3 3H9a3 3 0 0 1-3-3v-5zm3 3h.01M15 13h.01", + /** Skill library (an open book: two pages + spine). */ + skills: "M2 3h6a4 4 0 0 1 4 4v14a3 3 0 0 0-3-3H2zM22 3h-6a4 4 0 0 0-4 4v14a3 3 0 0 1 3-3h7z", + models: "M7 7h10v10H7zM4 10h3m10 0h3M4 14h3m10 0h3M10 4v3m4-3v3m-4 10v3m4-3v3", + usage: "M4 20V10m6 10V4m6 16v-7m4 7H2", + traces: "M4 6h16M4 12h10M4 18h13", + /** Benchmark center (a trophy: cup + two handles + base). */ + benchmark: + "M7 4h10v5a5 5 0 0 1-10 0V4zM7 5H4v1a3 3 0 0 0 3 3m10-4h3v1a3 3 0 0 1-3 3M12 14v4m-4 0h8", +} as const; + +/** Standard gear (lucide settings): full tooth outline + center circle, crisp and undistorted at 16px. */ +const GEAR_ICON = + "M12.22 2h-.44a2 2 0 0 0-2 2v.18a2 2 0 0 1-1 1.73l-.43.25a2 2 0 0 1-2 0l-.15-.08a2 2 0 0 0-2.73.73l-.22.38a2 2 0 0 0 .73 2.73l.15.1a2 2 0 0 1 1 1.72v.51a2 2 0 0 1-1 1.74l-.15.09a2 2 0 0 0-.73 2.73l.22.38a2 2 0 0 0 2.73.73l.15-.08a2 2 0 0 1 2 0l.43.25a2 2 0 0 1 1 1.73V20a2 2 0 0 0 2 2h.44a2 2 0 0 0 2-2v-.18a2 2 0 0 1 1-1.73l.43-.25a2 2 0 0 1 2 0l.15.08a2 2 0 0 0 2.73-.73l.22-.39a2 2 0 0 0-.73-2.73l-.15-.08a2 2 0 0 1-1-1.74v-.5a2 2 0 0 1 1-1.74l.15-.09a2 2 0 0 0 .73-2.73l-.22-.38a2 2 0 0 0-2.73-.73l-.15.08a2 2 0 0 1-2 0l-.43-.25a2 2 0 0 1-1-1.73V4a2 2 0 0 0-2-2zM15 12a3 3 0 1 1-6 0 3 3 0 0 1 6 0z"; + +const menuItemClass = + "block w-full px-3.5 py-2 text-left text-sm transition-colors duration-150 hover:bg-gray-100 dark:hover:bg-gray-800"; + +/** Session status dot: running pulses green, compacting shows an amber dot; idle shows nothing. */ +function StatusDot({ session }: { session: SessionInfo }) { + if (session.status === "running") { + return ( + + ); + } + if (session.status === "compacting") { + return ( + + ); + } + return null; +} + +export function Sidebar({ + onNavigate, + onCollapse, +}: { + onNavigate?: () => void; + onCollapse?: () => void; +}) { + const navigate = useNavigate(); + const { user, logout } = useAuth(); + const { mode, setMode, fontScale, setFontScale, accent, setAccent, currency, setCurrency } = + useTheme(); + const { lang, setLang } = useLocale(); + const { + projects, + currentProject, + setCurrentProjectId, + reloadProjects, + agents, + setCurrentAgentId, + } = useProject(); + const { byAgent, loading, remove, replace } = useSessions(); + const chatMatch = useMatch("/chat/:sessionId"); + const activeSessionId = chatMatch?.params.sessionId ?? null; + + const [projectOpen, setProjectOpen] = useState(false); + const [userOpen, setUserOpen] = useState(false); + const [createProjectOpen, setCreateProjectOpen] = useState(false); + const [projectSettingsOpen, setProjectSettingsOpen] = useState(false); + const [changePasswordOpen, setChangePasswordOpen] = useState(false); + /** Collapsed Agent groups (expanded by default). */ + const [collapsedAgents, setCollapsedAgents] = useState>(new Set()); + /** Expanded "archived" groups (collapsed by default). */ + const [openArchived, setOpenArchived] = useState>(new Set()); + /** Session pending delete confirmation (null = none). */ + const [deletingSession, setDeletingSession] = useState(null); + const [deletingBusy, setDeletingBusy] = useState(false); + const [deleteError, setDeleteError] = useState(null); + /** Session currently being renamed (null = none) and the title being typed. */ + const [renamingSession, setRenamingSession] = useState(null); + const [renameText, setRenameText] = useState(""); + const [renameBusy, setRenameBusy] = useState(false); + const [renameError, setRenameError] = useState(null); + + const toggleAgent = (agentId: string) => + setCollapsedAgents((prev) => { + const next = new Set(prev); + if (next.has(agentId)) next.delete(agentId); + else next.add(agentId); + return next; + }); + + const toggleArchivedGroup = (agentId: string) => + setOpenArchived((prev) => { + const next = new Set(prev); + if (next.has(agentId)) next.delete(agentId); + else next.add(agentId); + return next; + }); + + /** Archive / unarchive: persists immediately and updates in place (fails silently; the next list refresh self-corrects). */ + const toggleArchive = async (s: SessionInfo) => { + // Archiving the currently open chat: expand the "archived" group so it doesn't silently vanish from the sidebar with no way back. + if (!s.archived && s.sessionId === activeSessionId) { + setOpenArchived((prev) => new Set(prev).add(s.agentId)); + } + try { + const res = await api.patchSession(s.sessionId, { archived: !s.archived }); + replace(res.session); + } catch { + /* Ignore: non-critical operation */ + } + }; + + const confirmRename = async () => { + if (!renamingSession) return; + const title = renameText.trim(); + if (!title) return; + setRenameBusy(true); + setRenameError(null); + try { + const res = await api.patchSession(renamingSession.sessionId, { title }); + replace(res.session); + setRenamingSession(null); + } catch (e) { + setRenameError(e instanceof ApiError ? e.message : S.common.unknownError); + } finally { + setRenameBusy(false); + } + }; + + const confirmDeleteSession = async () => { + if (!deletingSession) return; + setDeletingBusy(true); + setDeleteError(null); + const target = deletingSession; + try { + await api.deleteSession(target.sessionId); + remove(target.sessionId); + // The session is gone, so clear its input draft too (no orphaned keys left in localStorage; keys are scoped per user, #68). + if (user) clearDraft(sessionDraftKey(user.userId, target.sessionId)); + setDeletingSession(null); + // The deleted session was the one open: jump to another **unarchived** Session in the same + // group, otherwise fall back to the chat home page (never jump into an archived session — + // it's hidden by default, so landing there would look like the chat vanished into thin air). + if (activeSessionId === target.sessionId) { + const rest = (byAgent.get(target.agentId) ?? []).filter( + (s) => s.sessionId !== target.sessionId && !s.archived, + ); + navigate(rest[0] ? `/chat/${rest[0].sessionId}` : "/chat"); + } + } catch (e) { + setDeleteError(e instanceof ApiError ? e.message : S.common.unknownError); + } finally { + setDeletingBusy(false); + } + }; + + const go = (to: string) => { + navigate(to); + onNavigate?.(); + }; + + /** + * New chat: enters draft state (/chat/new) without creating a Session — Model / Workspace / + * approval mode are all chosen on the draft input card, and the Session is only actually + * created when the first message is sent. The route state explicitly carries the target + * Agent: the group header's "+" uses that group's Agent, while the menu's "New chat" uses + * default_agent; this explicit intent overrides the previously selected Agent in the draft + * cache (the rest of the draft content, such as the message body, is preserved). + */ + const newChat = (agentId?: string) => { + if (agentId) setCurrentAgentId(agentId); + navigate(`/chat/${DRAFT_SESSION_ID}`, agentId ? { state: { agentId } } : undefined); + onNavigate?.(); + }; + + /** Target of the menu's "New chat": default_agent, falling back to the first Agent (if the list isn't ready yet, resolution is deferred to the draft page). */ + const defaultAgentId = (agents.find((a) => a.agentId === "default_agent") ?? agents[0])?.agentId; + + const openSession = (s: SessionInfo) => { + // Cross-group click: the current Agent follows this Session's own Agent. + setCurrentAgentId(s.agentId); + go(`/chat/${s.sessionId}`); + }; + + const navItems: Array<{ to: string; label: string; icon: string }> = [ + { to: "/agents", label: S.nav.agents, icon: NAV_ICONS.agents }, + { to: "/skills", label: S.nav.skills, icon: NAV_ICONS.skills }, + { to: "/models", label: S.nav.models, icon: NAV_ICONS.models }, + { to: "/usage", label: S.nav.usage, icon: NAV_ICONS.usage }, + { to: "/traces", label: S.nav.traces, icon: NAV_ICONS.traces }, + { to: "/benchmark", label: S.nav.benchmark, icon: NAV_ICONS.benchmark }, + ]; + + const themeOptions: ReadonlyArray<{ value: ThemeMode; label: string }> = [ + { value: "light", label: S.settings.themeLight }, + { value: "dark", label: S.settings.themeDark }, + { value: "system", label: S.settings.followSystem }, + ]; + const langOptions: ReadonlyArray<{ value: LangPref; label: string }> = [ + { value: "en", label: S.settings.langEn }, + { value: "zh", label: S.settings.langZh }, + { value: "system", label: S.settings.followSystem }, + ]; + const fontOptions: ReadonlyArray<{ value: FontScale; label: string }> = [ + { value: "sm", label: S.settings.fontSmall }, + { value: "md", label: S.settings.fontMedium }, + { value: "lg", label: S.settings.fontLarge }, + ]; + const currencyOptions: ReadonlyArray<{ value: Currency; label: string }> = [ + { value: "USD", label: S.models.currencyUsd }, + { value: "CNY", label: S.models.currencyCny }, + ]; + + return ( +
    + {/* Project switcher (+ collapse sidebar) */} +
    + {onCollapse && ( + + )} + setProjectOpen(!projectOpen)} + className="flex w-full items-center gap-1.5 rounded-md px-2 py-1.5 text-base font-semibold transition-colors duration-150 hover:bg-gray-200/70 dark:hover:bg-gray-800" + > + + {currentProject ? projectDisplayName(currentProject) : S.common.loading} + + + + + + } + > + {projects.map((p) => ( + + ))} +
    + + {currentProject && ( + + )} +
    +
    +
    + + {/* Fixed nav (new chat pinned at top: default_agent draft): no background fill, shares the + same gray hover/active styling as nav items, distinguished only by its top position and + font-medium; shows the same gray active state while on the draft page. */} + + + {/* Session area grouped by Agent (scrollable) */} +
    + {loading && agents.length === 0 ? ( + + ) : ( + agents.map((agent) => { + const list = byAgent.get(agent.agentId) ?? []; + const activeList = list.filter((s) => !s.archived); + const archivedList = list.filter((s) => s.archived); + const collapsed = collapsedAgents.has(agent.agentId); + const archivedOpen = openArchived.has(agent.agentId); + return ( +
    + {/* Group header: collapse toggle (Agent name) + new chat + Agent settings */} +
    + + {/* New chat: enters draft state directly with this group's Agent (all options live on the draft input card) */} + + +
    + + {collapsed ? null : ( + <> + {activeList.length === 0 && archivedList.length === 0 ? ( +

    + {S.chat.noSessions} +

    + ) : ( +
      + {activeList.map((s) => ( + { + setRenameError(null); + setRenameText(x.title ?? ""); + setRenamingSession(x); + }} + onDelete={(x) => { + setDeleteError(null); + setDeletingSession(x); + }} + onToggleArchive={(x) => void toggleArchive(x)} + /> + ))} +
    + )} + + {/* Archived group (collapsed by default) */} + {archivedList.length > 0 && ( +
    + + {archivedOpen && ( +
      + {archivedList.map((s) => ( + { + setRenameError(null); + setRenameText(x.title ?? ""); + setRenamingSession(x); + }} + onDelete={(x) => { + setDeleteError(null); + setDeletingSession(x); + }} + onToggleArchive={(x) => void toggleArchive(x)} + /> + ))} +
    + )} +
    + )} + + )} +
    + ); + }) + )} +
    + + {/* Bottom user config */} +
    + setUserOpen(!userOpen)} + className="flex w-full items-center gap-2 rounded-md px-2 py-1.5 text-left transition-colors duration-150 hover:bg-gray-200/70 dark:hover:bg-gray-800" + > + + {(user?.userId ?? "?").slice(0, 1).toUpperCase()} + + {user?.userId} + {user?.isAdmin && ( + {S.auth.admin} + )} + + } + > +
    + + + + + + + + + + + + + + + +
    +
    + + {/* User management is visible only to admins (the page route also has its own guard as a fallback). */} + {user?.isAdmin && ( + + )} + +
    +
    +
    + + setChangePasswordOpen(false)} + /> + + setCreateProjectOpen(false)} + onCreated={(projectId) => { + setCreateProjectOpen(false); + void reloadProjects().then(() => setCurrentProjectId(projectId)); + }} + /> + {currentProject && ( + setProjectSettingsOpen(false)} + /> + )} + {/* Rename chat */} + (renameBusy ? undefined : setRenamingSession(null))} + footer={ + <> + + + + } + > + setRenameText(e.target.value)} + onKeyDown={(e) => { + if (e.key === "Enter" && renameText.trim() && !renameBusy) void confirmRename(); + }} + /> + {renameError && ( +

    {renameError}

    + )} +
    + + {/* Delete chat confirmation */} + (deletingBusy ? undefined : setDeletingSession(null))} + footer={ + <> + + + + } + > +

    + {deletingSession + ? S.chat.deleteSessionConfirm(deletingSession.title ?? S.chat.defaultSessionTitle) + : ""} +

    + {deleteError && ( +

    {deleteError}

    + )} +
    +
    + ); +} + +/** Single Session row: title + status dot/approval badge + hover action group (rename, archive/unarchive, delete). */ +function SessionRow({ + s, + active, + onOpen, + onRename, + onDelete, + onToggleArchive, +}: { + s: SessionInfo; + active: boolean; + onOpen: (s: SessionInfo) => void; + onRename: (s: SessionInfo) => void; + onDelete: (s: SessionInfo) => void; + onToggleArchive: (s: SessionInfo) => void; +}) { + const actionBtn = + "flex h-6 w-6 shrink-0 items-center justify-center rounded text-gray-400 opacity-0 transition-all duration-150 focus-visible:opacity-100 group-hover:opacity-100"; + return ( +
  • +
    + + {/* Action group: rename + archive/unarchive + delete */} +
    + + + +
    +
    +
  • + ); +} + +function SettingRow({ label, children }: { label: string; children: ReactNode }) { + return ( +
    +

    {label}

    + {children} +
    + ); +} + +/** Accent color picker: a row of swatches, with a ring on the selected one. */ +function AccentPicker({ value, onChange }: { value: Accent; onChange: (a: Accent) => void }) { + return ( +
    + {ACCENT_SWATCHES.map((s) => ( +
    + ); +} diff --git a/packages/web/src/components/ui/agent-avatar.tsx b/packages/web/src/components/ui/agent-avatar.tsx new file mode 100644 index 0000000..e7ffbbe --- /dev/null +++ b/packages/web/src/components/ui/agent-avatar.tsx @@ -0,0 +1,90 @@ +/** + * Agent avatar: a pixel identicon deterministically generated from agentId (the same approach + * GitHub's default avatars use). + * + * 5x5 grid with left-right mirror symmetry (only the left 3 columns are randomized, the right 2 + * columns mirror back), a single soft-hue foreground, and a very light background of the same + * color. No external dependencies — seeds mulberry32 with an FNV-1a hash. + */ + +/** Grid side length (must be odd for left-right symmetry). */ +const N = 5; +/** Number of columns that need randomizing (including the center column). */ +const HALF = Math.ceil(N / 2); + +function hashStr(s: string): number { + let h = 2166136261; + for (let i = 0; i < s.length; i++) { + h ^= s.charCodeAt(i); + h = Math.imul(h, 16777619); + } + return h >>> 0; +} + +function mulberry32(a: number): () => number { + return () => { + a |= 0; + a = (a + 0x6d2b79f5) | 0; + let t = Math.imul(a ^ (a >>> 15), 1 | a); + t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t; + return ((t ^ (t >>> 14)) >>> 0) / 4294967296; + }; +} + +/** Randomly samples points on the left half + center column, mirrored into a full 5x5 boolean grid. */ +function buildGrid(rnd: () => number): boolean[][] { + const grid: boolean[][] = Array.from({ length: N }, () => Array.from({ length: N }, () => false)); + for (let col = 0; col < HALF; col++) { + for (let row = 0; row < N; row++) { + // The center column has a slightly lower fill rate, to avoid a solid vertical line. + const on = rnd() < (col === HALF - 1 ? 0.4 : 0.55); + grid[row]![col] = on; + grid[row]![N - 1 - col] = on; + } + } + return grid; +} + +export function AgentAvatar({ + id, + size = 18, + className, +}: { + id: string; + size?: number; + className?: string; +}) { + const rnd = mulberry32(hashStr(id || "agent")); + const hue = Math.floor(rnd() * 360); + const grid = buildGrid(rnd); + // 24x24 viewport: 2px margin on each side, 4x4 cells. + const cell = 4; + const pad = 2; + const fg = `hsl(${hue} 52% 46%)`; + return ( + + + {grid.map((cols, row) => + cols.map((on, col) => + on ? ( + + ) : null, + ), + )} + + ); +} diff --git a/packages/web/src/components/ui/badge.tsx b/packages/web/src/components/ui/badge.tsx new file mode 100644 index 0000000..8eda564 --- /dev/null +++ b/packages/web/src/components/ui/badge.tsx @@ -0,0 +1,36 @@ +/** + * Badge component: a small pill-shaped label for status/type (stop_reason, running status, Trace event type, etc.). + */ +import type { ReactNode } from "react"; + +export type BadgeTone = "gray" | "brand" | "green" | "amber" | "red"; + +const toneClass: Record = { + gray: "bg-gray-100 text-gray-600 dark:bg-gray-800 dark:text-gray-300", + brand: "bg-gray-200/80 text-gray-700 dark:bg-gray-700/60 dark:text-gray-200", + green: "bg-emerald-50 text-emerald-700 dark:bg-emerald-950 dark:text-emerald-300", + amber: "bg-amber-50 text-amber-700 dark:bg-amber-950 dark:text-amber-300", + red: "bg-red-50 text-red-700 dark:bg-red-950 dark:text-red-300", +}; + +export function Badge({ tone = "gray", children }: { tone?: BadgeTone; children: ReactNode }) { + return ( + + {children} + + ); +} + +/** stop_reason -> badge tone (completed usually shows no badge). */ +export function stopReasonTone(stopReason: string): BadgeTone { + switch (stopReason) { + case "completed": + return "green"; + case "aborted": + return "amber"; + default: + return "red"; // failed / timeout / malformed + } +} diff --git a/packages/web/src/components/ui/button.tsx b/packages/web/src/components/ui/button.tsx new file mode 100644 index 0000000..f670e90 --- /dev/null +++ b/packages/web/src/components/ui/button.tsx @@ -0,0 +1,45 @@ +/** + * Button component: GitHub-style simplicity — small border radius + 1px border + a single brand accent, only color transitions. + */ +import type { ButtonHTMLAttributes } from "react"; + +type Variant = "primary" | "secondary" | "danger" | "ghost"; +type Size = "sm" | "md" | "icon"; + +const variantClass: Record = { + // primary uses the theme accent variable (defaults to neutral gray/white, switching with light/dark; becomes that color once an accent is selected). + primary: + "bg-[var(--accent-bg)] text-[var(--accent-fg)] border border-[var(--accent-bg)] " + + "transition-opacity hover:opacity-90 disabled:opacity-50", + secondary: + "bg-white text-gray-800 border border-gray-300 hover:bg-gray-50 " + + "dark:bg-gray-900 dark:text-gray-200 dark:border-gray-700 dark:hover:bg-gray-800", + danger: + "bg-white text-red-600 border border-gray-300 hover:border-red-300 hover:bg-red-50 " + + "dark:bg-gray-900 dark:text-red-400 dark:border-gray-700 dark:hover:bg-red-950", + ghost: + "bg-transparent text-gray-600 border border-transparent hover:bg-gray-100 hover:text-gray-900 " + + "dark:text-gray-300 dark:hover:bg-gray-800 dark:hover:text-gray-100", +}; + +const sizeClass: Record = { + sm: "px-2.5 py-1 text-xs rounded-md", + md: "px-3 py-1.5 text-sm rounded-md", + /** Square icon button (no text; callers must supply title / aria-label). */ + icon: "p-1.5 rounded-md", +}; + +export interface ButtonProps extends ButtonHTMLAttributes { + variant?: Variant; + size?: Size; +} + +export function Button({ variant = "secondary", size = "md", className, ...rest }: ButtonProps) { + return ( + + +
    + {children} +
    + + + ); +} diff --git a/packages/web/src/components/ui/dropdown.tsx b/packages/web/src/components/ui/dropdown.tsx new file mode 100644 index 0000000..adb79bd --- /dev/null +++ b/packages/web/src/components/ui/dropdown.tsx @@ -0,0 +1,56 @@ +/** + * Dropdown container (controlled): clicking outside collapses it; the panel is absolutely + * positioned (z-40, per the layering convention — chrome avoids stacking contexts, menus are + * z-40, overlays are z-50). menuClass controls the docking direction and width. + */ +import { useEffect, useRef } from "react"; +import type { ReactNode } from "react"; + +export function Dropdown({ + button, + open, + setOpen, + children, + menuClass, + className, +}: { + button: ReactNode; + open: boolean; + setOpen: (v: boolean) => void; + children: ReactNode; + /** Panel positioning and size (default: downward, left-aligned, w-64). */ + menuClass?: string; + /** Extra classes for the root container (e.g. flex-1 in a flex layout). */ + className?: string; +}) { + const ref = useRef(null); + useEffect(() => { + if (!open) return; + const onClick = (e: MouseEvent) => { + if (ref.current && !ref.current.contains(e.target as Node)) setOpen(false); + }; + const onKey = (e: KeyboardEvent) => { + if (e.key === "Escape") setOpen(false); + }; + window.addEventListener("mousedown", onClick); + window.addEventListener("keydown", onKey); + return () => { + window.removeEventListener("mousedown", onClick); + window.removeEventListener("keydown", onKey); + }; + }, [open, setOpen]); + return ( +
    + {button} + {open && ( +
    + {children} +
    + )} +
    + ); +} diff --git a/packages/web/src/components/ui/empty-state.tsx b/packages/web/src/components/ui/empty-state.tsx new file mode 100644 index 0000000..fe53c65 --- /dev/null +++ b/packages/web/src/components/ui/empty-state.tsx @@ -0,0 +1,22 @@ +/** + * Empty state component: a placeholder message for when a list/detail view has no data (plain text, no graphic decoration). + */ +import type { ReactNode } from "react"; + +export function EmptyState({ + title, + description, + action, +}: { + title: string; + description?: string; + action?: ReactNode; +}) { + return ( +
    +

    {title}

    + {description &&

    {description}

    } + {action &&
    {action}
    } +
    + ); +} diff --git a/packages/web/src/components/ui/glyph-icon.tsx b/packages/web/src/components/ui/glyph-icon.tsx new file mode 100644 index 0000000..e71ae37 --- /dev/null +++ b/packages/web/src/components/ui/glyph-icon.tsx @@ -0,0 +1,31 @@ +/** + * Unified rendering for stat icons (shared by the chat page's stat row and the Trace page's turn + * cards): a 24x24 line path with stroke set to currentColor, so the color follows the caller's + * text color. See lib/stat-icons.ts for the paths. + */ +export function GlyphIcon({ + d, + size = 13, + className = "", +}: { + d: string; + size?: number; + className?: string; +}) { + return ( + + + + ); +} diff --git a/packages/web/src/components/ui/image-zoom.tsx b/packages/web/src/components/ui/image-zoom.tsx new file mode 100644 index 0000000..db0922a --- /dev/null +++ b/packages/web/src/components/ui/image-zoom.tsx @@ -0,0 +1,72 @@ +/** + * Clickable image that zooms in on click. The thumbnail keeps the caller's styling; + * clicking it opens a lightbox: a bordered image panel with a close glyph in the + * top-right corner (no title bar), closable via Esc or clicking the overlay. + * The lightbox is rendered via portal to body — the thumbnail may be nested inside + * a card with a transform entrance animation or overflow-hidden, and a `fixed` + * layer rendered in place would get hijacked/clipped by that ancestor (same fix + * as the Select dropdown). + */ +import { useEffect, useState } from "react"; +import { createPortal } from "react-dom"; +import { S } from "../../lib/strings"; + +export function ZoomableImage({ + src, + alt, + className, +}: { + src: string; + alt: string; + /** Style for the thumbnail img (keeps the caller's original class). */ + className?: string; +}) { + const [open, setOpen] = useState(false); + return ( + <> + + {open && setOpen(false)} />} + + ); +} + +function Lightbox({ src, alt, onClose }: { src: string; alt: string; onClose: () => void }) { + useEffect(() => { + const onKey = (e: KeyboardEvent) => { + if (e.key === "Escape") onClose(); + }; + window.addEventListener("keydown", onKey); + return () => window.removeEventListener("keydown", onKey); + }, [onClose]); + + return createPortal( +
    { + if (e.target === e.currentTarget) onClose(); + }} + > +
    + {/* Close glyph: top-right inside the frame, floating over the image (dark semi-transparent background keeps it visible on any image). */} + + {alt} +
    +
    , + document.body, + ); +} diff --git a/packages/web/src/components/ui/input.tsx b/packages/web/src/components/ui/input.tsx new file mode 100644 index 0000000..a14ca80 --- /dev/null +++ b/packages/web/src/components/ui/input.tsx @@ -0,0 +1,118 @@ +/** + * Text input component: optional label and hint/error text; rounded corners with + * a hover border darken and brand focus-ring transition. + */ +import { forwardRef } from "react"; +import type { InputHTMLAttributes, TextareaHTMLAttributes } from "react"; + +// Excludes font size and padding: each of Input/Textarea appends its own (see their size). +const baseClass = + "w-full rounded-md border border-gray-300 bg-white text-gray-900 " + + "placeholder:text-gray-400 transition-[border-color,box-shadow] duration-200 " + + "hover:border-gray-400 focus:border-gray-500 focus:outline-none focus:ring-2 focus:ring-gray-400/30 " + + "disabled:cursor-not-allowed disabled:opacity-60 " + + "dark:border-gray-700 dark:bg-gray-900 dark:text-gray-100 dark:placeholder:text-gray-500 " + + "dark:hover:border-gray-600 dark:focus:border-gray-400 dark:focus:ring-gray-500/30"; + +/** Size tier: base (form default) / sm (compact contexts like filter bars, keeps the toolbar from growing taller). */ +export type ControlSize = "base" | "sm"; + +export const sizeClass: Record = { + base: "px-3 py-2 text-base", + sm: "px-2 py-1 text-xs", +}; + +export interface InputProps extends Omit, "size"> { + label?: string; + hint?: string; + error?: string; + /** + * Marks the field red without rendering error text: use this when the input has a + * custom wrapper (prefix glyph, unit suffix, etc.) and the caller places the error + * text outside that wrapper — otherwise the text would get pulled into the + * absolutely-positioned reference frame and skew the prefix/suffix layout. + */ + invalid?: boolean; + size?: ControlSize; +} + +/** + * Error state: red border + light red background (a failed field is visible at a + * glance; the error text sits below the box). + * Forced with `!` — baseClass's border-gray-300 / bg-white are the same kind of + * border/background utility classes, and which one wins depends on the order the + * CSS was generated in, not the order of classes in the string (without `!important` + * this would get overridden). + */ +const errorClass = + "!border-red-400 !bg-red-50 hover:!border-red-500 focus:!border-red-500 focus:!ring-red-400/30 " + + "dark:!border-red-800 dark:!bg-red-950/40 dark:hover:!border-red-700 dark:focus:!border-red-600"; + +export function Input({ + label, + hint, + error, + invalid, + size = "base", + className, + ...rest +}: InputProps) { + const bad = Boolean(error) || Boolean(invalid); + const control = ( + + ); + if (!label && !hint && !error) return control; + return ( + + ); +} + +export interface TextareaProps extends TextareaHTMLAttributes { + label?: string; + hint?: string; + /** Monospace font (for editing Prompts/parameters). */ + mono?: boolean; + /** Font size: base (default, matches body text) or sm (smaller, for editing long Prompts). */ + size?: "base" | "sm"; +} + +export const Textarea = forwardRef(function Textarea( + { label, hint, mono, size = "base", className, ...rest }, + ref, +) { + const control = ( +