Initialize repository with harness code and assets

Initial import of all source code, config, and README assets: the
packages workspace (cli, core, server, web, docs, landing, skills),
build scripts, tooling config, and CI workflows.

Includes the data-layout revision made on this branch: the local data
root defaults to ~/.penguin/data (PENGUIN_HOME still overrides; the
installer keeps its binaries in ~/.penguin), and every Agent lives
under <project>/agents/<agent>/ — path helpers, the three
agent-enumeration scans, the system prompt, built-in Skills, tests
and docs all follow the new layout.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_018ihk8iQuo3kv2aPjAYEPuR
This commit is contained in:
Yaowei Zheng
2026-07-19 14:06:53 +08:00
committed by GitHub
parent 056bed7aeb
commit 45bfae6e94
543 changed files with 92949 additions and 0 deletions
@@ -0,0 +1,96 @@
/**
* Admin user backend: user list / create / reset password / delete.
*
* - Create: username is the user_id (^[a-z][a-z0-9_-]{1,31}$); admin sets the initial
* password, flagged with password_is_initial; a default Project `proj-<username>` is
* auto-created, rolling back the user row on failure.
* - Reset password: also flags the password as initial and clears all of the user's
* login sessions (forcing re-login).
* - Delete: the built-in admin cannot be deleted; Projects owned by the user are
* deleted along with it (including data directories), with sessions/memberships/UI
* preferences cascade-deleted via foreign keys.
*/
import type { UserInfo } from "../api/types.js";
import { HttpError } from "../http/errors.js";
import { MIN_PASSWORD_LENGTH, toUserInfo } from "../auth/service.js";
import { hashPassword } from "../auth/password.js";
import type { AuthSessionsRepo } from "../db/repos/auth-sessions.js";
import type { ProjectsRepo } from "../db/repos/projects.js";
import type { UserRow, UsersRepo } from "../db/repos/users.js";
import { SEMANTIC_ID_RULE, USERNAME_PATTERN } from "./ids.js";
import type { ProjectService } from "./project-service.js";
export interface AdminServiceDeps {
users: UsersRepo;
authSessions: AuthSessionsRepo;
projects: ProjectsRepo;
projectService: ProjectService;
now?: () => Date;
}
export class AdminService {
private readonly now: () => Date;
constructor(private readonly deps: AdminServiceDeps) {
this.now = deps.now ?? (() => new Date());
}
listUsers(): UserInfo[] {
return this.deps.users.list().map(toUserInfo);
}
async createUser(userId: string, password: string): Promise<UserInfo> {
if (!USERNAME_PATTERN.test(userId)) {
throw new HttpError(400, "invalid_user_id", `用户名须为 2~32 位:${SEMANTIC_ID_RULE}。`);
}
if (password.length < MIN_PASSWORD_LENGTH) {
throw new HttpError(400, "invalid_password", "密码至少 8 个字符。");
}
if (this.deps.users.findById(userId)) {
throw new HttpError(409, "user_exists", `用户已存在:${userId}。`);
}
const user: UserRow = {
userId,
passwordHash: await hashPassword(password),
isAdmin: false,
passwordIsInitial: true,
createdAt: this.now().toISOString(),
};
this.deps.users.insert(user);
try {
await this.deps.projectService.provisionInitialProject(user, false);
} catch (err) {
// Compensation: roll back the user row if default Project creation fails (e.g. proj-<username> already taken).
this.deps.users.delete(user.userId);
throw err;
}
return toUserInfo(user);
}
/** Reset another user's password: flags it as initial and clears all their sessions (prompts a password change on next login). */
async resetPassword(userId: string, password: string): Promise<void> {
if (!this.deps.users.findById(userId)) {
throw new HttpError(404, "user_not_found", `用户不存在:${userId}。`);
}
if (password.length < MIN_PASSWORD_LENGTH) {
throw new HttpError(400, "invalid_password", "密码至少 8 个字符。");
}
this.deps.users.updatePassword(userId, await hashPassword(password), true);
this.deps.authSessions.deleteByUser(userId);
}
/** Delete user: the built-in admin cannot be deleted; owned Projects (including data directories) are deleted along with it. */
async deleteUser(userId: string): Promise<void> {
const target = this.deps.users.findById(userId);
if (!target) {
throw new HttpError(404, "user_not_found", `用户不存在:${userId}。`);
}
if (target.isAdmin) {
throw new HttpError(409, "cannot_delete_admin", "内置管理员不能删除。");
}
for (const project of this.deps.projects.listByOwner(userId)) {
await this.deps.projectService.destroyProject(project.projectId);
}
this.deps.users.delete(userId); // auth_sessions / project_members / ui_prefs cascade-deleted
}
}
@@ -0,0 +1,341 @@
/**
* Agent config read/write (config is an editable file).
*
* system_config.yaml is edited via yaml's `parseDocument`: only the keys provided in
* the request are updated, the rest of the file (including comments) is preserved
* as-is; AGENTS.md is overwritten in full.
* The vault (agent_state/.vault.toml) is read/written via core's loadAgentVault/saveAgentVault;
* plaintext values only ever hit disk, and are always masked in responses.
*/
import fs from "node:fs/promises";
import { parseDocument, parse as parseYaml } from "yaml";
import {
agentsMdPath,
agentStateDir,
agentStateVersion,
VAULT_VALUE_MAX_LENGTH,
isValidVaultKey,
loadAgentVault,
saveAgentVault,
systemConfigPath,
} from "@prismshadow/penguin-core";
import type {
MCPServerConfig,
ThinkingLevelName,
ToolDefinitionConfig,
} from "@prismshadow/penguin-core";
import type {
AgentConfigDto,
AgentConfigUpdateRequest,
AgentModelConfigDto,
AgentCompactionConfigDto,
VaultEntryInfo,
VaultResponse,
VaultUpdateRequest,
} from "../api/types.js";
import { HttpError } from "../http/errors.js";
import { badRequest, optionalEnum, optionalNumber, optionalString } from "../http/validate.js";
import { maskApiKey } from "./project-config-service.js";
const THINKING_LEVELS: readonly ThinkingLevelName[] = ["none", "low", "medium", "high", "xhigh"];
const COMPACTION_MODES = ["summarize", "discard"] as const;
function asRecord(v: unknown): Record<string, unknown> {
return v !== null && typeof v === "object" && !Array.isArray(v)
? (v as Record<string, unknown>)
: {};
}
export interface AgentConfigView {
agentsMd: string;
systemConfigYaml: string;
config: AgentConfigDto;
stateDir: string;
}
export class AgentConfigService {
constructor(private readonly root: string) {}
/** Whether the Agent exists (determined by the presence of system_config.yaml, matching the CLI's convention). */
async exists(projectId: string, agentId: string): Promise<boolean> {
try {
await fs.access(systemConfigPath(this.root, projectId, agentId));
return true;
} catch {
return false;
}
}
async requireExists(projectId: string, agentId: string): Promise<void> {
if (!(await this.exists(projectId, agentId))) {
throw new HttpError(404, "agent_not_found", "Agent 不存在。");
}
}
/**
* Read list-card metadata: name / description + tool count (sum of tools.builtin
* and tools.mcpServers entries; MCP counted per server). Silently falls back to
* empty / 0 if the file is corrupt.
*/
async readCardMeta(
projectId: string,
agentId: string,
): Promise<{ name?: string; description?: string; toolCount: number; version: number }> {
try {
const raw = await fs.readFile(systemConfigPath(this.root, projectId, agentId), "utf8");
const parsed = asRecord(parseYaml(raw));
const tools = asRecord(parsed.tools);
const countOf = (v: unknown): number => (Array.isArray(v) ? v.length : 0);
return {
...(typeof parsed.name === "string" ? { name: parsed.name } : {}),
...(typeof parsed.description === "string" ? { description: parsed.description } : {}),
toolCount: countOf(tools.builtin) + countOf(tools.mcpServers),
version: agentStateVersion({ version: parsed.version as number | undefined }),
};
} catch {
return { toolCount: 0, version: 1 };
}
}
/** Structured config view (matching the edit form's shape) + raw text + AGENTS.md + State path. */
async getConfig(projectId: string, agentId: string): Promise<AgentConfigView> {
await this.requireExists(projectId, agentId);
const yamlPath = systemConfigPath(this.root, projectId, agentId);
const systemConfigYaml = await fs.readFile(yamlPath, "utf8");
const parsed = asRecord(parseYaml(systemConfigYaml));
const model = asRecord(parsed.model);
const compaction = asRecord(parsed.compaction);
const tools = asRecord(parsed.tools);
let agentsMd = "";
try {
agentsMd = await fs.readFile(agentsMdPath(this.root, projectId, agentId), "utf8");
} catch {
// Treat a missing AGENTS.md as an empty file (it normally exists after initialization).
}
const modelDto: AgentModelConfigDto = {
...(typeof model.max_tokens === "number" ? { maxTokens: model.max_tokens } : {}),
...(typeof model.thinking_level === "string"
? { thinkingLevel: model.thinking_level as ThinkingLevelName }
: {}),
...(typeof model.timeoutMs === "number" ? { timeoutMs: model.timeoutMs } : {}),
};
const compactionDto: AgentCompactionConfigDto = {
...(typeof compaction.max_context_length === "number"
? { maxContextLength: compaction.max_context_length }
: {}),
...(typeof compaction.max_session_turns === "number"
? { maxSessionTurns: compaction.max_session_turns }
: {}),
...(compaction.mode === "summarize" || compaction.mode === "discard"
? { mode: compaction.mode }
: {}),
...(typeof compaction.prompt === "string" ? { prompt: compaction.prompt } : {}),
};
const config: AgentConfigDto = {
...(typeof parsed.name === "string" ? { name: parsed.name } : {}),
...(typeof parsed.description === "string" ? { description: parsed.description } : {}),
version: agentStateVersion({ version: parsed.version as number | undefined }),
systemPrompt: typeof parsed.system_prompt === "string" ? parsed.system_prompt : "",
...(typeof parsed.max_turns === "number" ? { maxTurns: parsed.max_turns } : {}),
...(Object.keys(modelDto).length > 0 ? { model: modelDto } : {}),
...(Object.keys(compactionDto).length > 0 ? { compaction: compactionDto } : {}),
toolsBuiltin: Array.isArray(tools.builtin) ? (tools.builtin as ToolDefinitionConfig[]) : [],
mcpServers: Array.isArray(tools.mcpServers) ? (tools.mcpServers as MCPServerConfig[]) : [],
};
return {
agentsMd,
systemConfigYaml,
config,
stateDir: agentStateDir(this.root, projectId, agentId),
};
}
/**
* PUT accepts any subset: only the provided keys are updated (parseDocument
* preserves comments and untouched content); agentsMd is overwritten in full.
* Numeric validation: >0 or -1; thinkingLevel / mode are validated as enums.
*/
async updateConfig(
projectId: string,
agentId: string,
req: AgentConfigUpdateRequest,
): Promise<void> {
await this.requireExists(projectId, agentId);
// Finish all config validation and document changes before writing to disk
// (if validation fails, AGENTS.md is not written either, avoiding a partial update).
if (req.config !== undefined) {
await this.applyConfigUpdate(projectId, agentId, req.config);
}
if (req.agentsMd !== undefined) {
await fs.writeFile(agentsMdPath(this.root, projectId, agentId), req.agentsMd, "utf8");
}
}
private async applyConfigUpdate(
projectId: string,
agentId: string,
config: NonNullable<AgentConfigUpdateRequest["config"]>,
): Promise<void> {
const cfg = config as unknown as Record<string, unknown>;
const yamlPath = systemConfigPath(this.root, projectId, agentId);
const doc = parseDocument(await fs.readFile(yamlPath, "utf8"));
const setIfProvided = (path: string[], value: unknown): void => {
if (value !== undefined) doc.setIn(path, value);
};
setIfProvided(["name"], optionalString(cfg, "name", { maxLen: 100, label: "name" }));
setIfProvided(
["description"],
optionalString(cfg, "description", { maxLen: 2000, label: "description" }),
);
setIfProvided(
["system_prompt"],
optionalString(cfg, "systemPrompt", { label: "systemPrompt" }),
);
setIfProvided(
["max_turns"],
optionalNumber(cfg, "maxTurns", { integer: true, positiveOrMinusOne: true }),
);
if (cfg.model !== undefined) {
const model = asRecord(cfg.model);
setIfProvided(
["model", "max_tokens"],
optionalNumber(model, "maxTokens", { integer: true, positiveOrMinusOne: true }),
);
setIfProvided(
["model", "thinking_level"],
optionalEnum(model, "thinkingLevel", THINKING_LEVELS),
);
setIfProvided(
["model", "timeoutMs"],
optionalNumber(model, "timeoutMs", { integer: true, positiveOrMinusOne: true }),
);
}
if (cfg.compaction !== undefined) {
const compaction = asRecord(cfg.compaction);
setIfProvided(
["compaction", "max_context_length"],
optionalNumber(compaction, "maxContextLength", { integer: true, positiveOrMinusOne: true }),
);
setIfProvided(
["compaction", "max_session_turns"],
optionalNumber(compaction, "maxSessionTurns", { integer: true, positiveOrMinusOne: true }),
);
setIfProvided(["compaction", "mode"], optionalEnum(compaction, "mode", COMPACTION_MODES));
setIfProvided(["compaction", "prompt"], optionalString(compaction, "prompt"));
}
if (cfg.toolsBuiltin !== undefined) {
doc.setIn(["tools", "builtin"], validateToolsBuiltin(cfg.toolsBuiltin));
}
if (cfg.mcpServers !== undefined) {
doc.setIn(["tools", "mcpServers"], validateMcpServers(cfg.mcpServers));
}
await fs.writeFile(yamlPath, doc.toString(), "utf8");
}
/** Read the Agent vault (agent_state/.vault.toml): values are always masked, plaintext is never sent to the client. */
async getVault(projectId: string, agentId: string): Promise<VaultResponse> {
await this.requireExists(projectId, agentId);
const vault = await loadAgentVault(this.root, projectId, agentId);
const entries: VaultEntryInfo[] = Object.entries(vault).map(([key, value]) => ({
key,
valueMasked: maskApiKey(value),
}));
return { entries };
}
/**
* PUT replaces the whole vault table (same semantics as models): keys absent from
* the body are deleted; omitting value keeps the existing value (a new key must
* provide a value). Key names are validated against shell environment variable
* naming rules (same rule as core); deleting everything removes the whole
* .vault.toml file.
*/
async updateVault(
projectId: string,
agentId: string,
req: VaultUpdateRequest,
): Promise<VaultResponse> {
await this.requireExists(projectId, agentId);
const prev = await loadAgentVault(this.root, projectId, agentId);
const seen = new Set<string>();
const nextVault: Record<string, string> = {};
for (const entry of req.entries) {
if (!isValidVaultKey(entry.key)) {
throw badRequest(
`vault 键名不合法:${entry.key}(仅字母、数字与下划线,且不能以数字开头)。`,
);
}
if (seen.has(entry.key)) {
throw badRequest(`entries 中存在重复的键名:${entry.key}。`);
}
seen.add(entry.key);
const prevValue = prev[entry.key];
if (entry.value !== undefined) {
// Values are injected into the child process environment: an oversized value would
// make exec spawn fail (E2BIG), so we reject it on write (same limit as core).
if (entry.value.length > VAULT_VALUE_MAX_LENGTH) {
throw badRequest(`vault 值过长:${entry.key}(上限 ${VAULT_VALUE_MAX_LENGTH} 字符)。`);
}
nextVault[entry.key] = entry.value;
} else if (prevValue !== undefined) {
nextVault[entry.key] = prevValue;
} else {
throw badRequest(`新增键 ${entry.key} 必须提供 value。`);
}
}
await saveAgentVault(this.root, projectId, agentId, nextVault);
return this.getVault(projectId, agentId);
}
}
function validateToolsBuiltin(value: unknown): ToolDefinitionConfig[] {
if (!Array.isArray(value)) throw badRequest("toolsBuiltin 必须是数组。");
return value.map((item, i) => {
const t = asRecord(item);
if (typeof t.name !== "string" || t.name.length === 0) {
throw badRequest(`toolsBuiltin[${i}].name 必须是非空字符串。`);
}
if (typeof t.description !== "string") {
throw badRequest(`toolsBuiltin[${i}].description 必须是字符串。`);
}
if (t.permission !== undefined && t.permission !== "r" && t.permission !== "rw") {
throw badRequest(`toolsBuiltin[${i}].permission 必须是 r / rw 之一。`);
}
if (t.forModel !== undefined && t.forModel !== "vision" && t.forModel !== "text-only") {
throw badRequest(`toolsBuiltin[${i}].forModel 必须是 vision / text-only 之一。`);
}
optionalNumber(t, "timeoutMs", {
integer: true,
positiveOrMinusOne: true,
label: `toolsBuiltin[${i}].timeoutMs`,
});
optionalNumber(t, "maxOutputLength", {
integer: true,
positiveOrMinusOne: true,
label: `toolsBuiltin[${i}].maxOutputLength`,
});
return t as unknown as ToolDefinitionConfig;
});
}
function validateMcpServers(value: unknown): MCPServerConfig[] {
if (!Array.isArray(value)) throw badRequest("mcpServers 必须是数组。");
return value.map((item, i) => {
const s = asRecord(item);
if (typeof s.name !== "string" || s.name.length === 0) {
throw badRequest(`mcpServers[${i}].name 必须是非空字符串。`);
}
if (s.config === null || typeof s.config !== "object" || Array.isArray(s.config)) {
throw badRequest(`mcpServers[${i}].config 必须是对象。`);
}
return s as unknown as MCPServerConfig;
});
}
@@ -0,0 +1,219 @@
/**
* Agent service.
*
* The list is the union of "DB index ∪ directory scan": a subdirectory under
* `<project>/` containing `agent_state/system_config.yaml` is treated as an Agent;
* unmanaged ones found are backfilled into the DB — this handles Agents created
* directly via the CLI.
* Create: generate agent-<8hex>, initialize Agent State via core's `createAgent`,
* then write name/description into system_config.yaml (parseDocument preserves the
* template's comments).
*/
import fs from "node:fs/promises";
import { HttpError } from "../http/errors.js";
import {
agentDir,
agentsDir,
agentsMdPath,
BUILTIN_AGENT_IDS,
createAgent as coreCreateAgent,
isValidId,
loadAgentVault,
scheduleDir,
systemConfigPath,
} from "@prismshadow/penguin-core";
import type { AgentsRepo } from "../db/repos/agents.js";
import { SEMANTIC_ID_PATTERN, SEMANTIC_ID_RULE } from "./ids.js";
import type { AgentConfigService } from "./agent-config-service.js";
export interface AgentListItem {
agentId: string;
name?: string;
description?: string;
createdAt?: string;
/** Last config modification time: the later of system_config.yaml / AGENTS.md mtime. */
updatedAt?: string;
/** Tool count: number of tools.builtin + tools.mcpServers entries (MCP counted per server). */
toolCount: number;
/** Agent State version number (missing field treated as 1). */
version: number;
/** Number of vault keys. */
vaultKeyCount: number;
/** Number of scheduled tasks (count of .toml files under schedule/, including invalid ones). */
scheduleCount: number;
}
export class AgentService {
constructor(
private readonly root: string,
private readonly agents: AgentsRepo,
private readonly agentConfig: AgentConfigService,
) {}
/** Union of DB index ∪ directory scan; unmanaged directory Agents are backfilled into the DB. */
async listAgents(projectId: string): Promise<AgentListItem[]> {
const known = new Map(this.agents.list(projectId).map((r) => [r.agentId, r]));
let entries: string[] = [];
try {
const dirents = await fs.readdir(agentsDir(this.root, projectId), { withFileTypes: true });
entries = dirents.filter((d) => d.isDirectory()).map((d) => d.name);
} catch {
// The Project's agents/ directory doesn't exist yet (no Agent directories): return from the DB index only.
}
for (const agentId of entries) {
if (known.has(agentId) || !isValidId(agentId)) continue;
const configPath = systemConfigPath(this.root, projectId, agentId);
let createdAt: string;
try {
const stat = await fs.stat(configPath);
createdAt = (stat.birthtime.getTime() > 0 ? stat.birthtime : stat.mtime).toISOString();
} catch {
continue; // A directory without system_config.yaml is not an Agent (e.g. a temp folder)
}
const row = { projectId, agentId, createdAt };
this.agents.insertOrIgnore(row);
known.set(agentId, row);
}
// Meta reads and mtime stats for each Agent run in parallel (Promise.all preserves the sorted order).
const sorted = [...known.values()].sort((a, b) =>
a.createdAt === b.createdAt
? a.agentId.localeCompare(b.agentId)
: a.createdAt < b.createdAt
? -1
: 1,
);
return Promise.all(
sorted.map(async (row) => {
const [meta, updatedAt, vaultKeyCount, scheduleCount] = await Promise.all([
this.agentConfig.readCardMeta(projectId, row.agentId),
this.configUpdatedAt(projectId, row.agentId),
this.vaultKeyCount(projectId, row.agentId),
this.scheduleCount(projectId, row.agentId),
]);
return {
agentId: row.agentId,
...meta,
createdAt: row.createdAt,
...(updatedAt !== undefined ? { updatedAt } : {}),
vaultKeyCount,
scheduleCount,
};
}),
);
}
/** Number of vault keys (falls back to 0 on read failure). */
private async vaultKeyCount(projectId: string, agentId: string): Promise<number> {
try {
return Object.keys(await loadAgentVault(this.root, projectId, agentId)).length;
} catch {
return 0;
}
}
/** Number of scheduled tasks: count of .toml files under schedule/ (0 if the directory doesn't exist). */
private async scheduleCount(projectId: string, agentId: string): Promise<number> {
try {
const names = await fs.readdir(scheduleDir(this.root, projectId, agentId));
return names.filter((n) => n.endsWith(".toml")).length;
} catch {
return 0;
}
}
/** Last config modification time: the later of system_config.yaml and AGENTS.md mtime; omitted if neither is readable. */
private async configUpdatedAt(projectId: string, agentId: string): Promise<string | undefined> {
const paths = [
systemConfigPath(this.root, projectId, agentId),
agentsMdPath(this.root, projectId, agentId),
];
const times = await Promise.all(
paths.map(async (p) => {
try {
return (await fs.stat(p)).mtime.getTime();
} catch {
return 0;
}
}),
);
const max = Math.max(...times);
return max > 0 ? new Date(max).toISOString() : undefined;
}
/**
* Delete an Agent: the sole built-in Agent
* default_agent (shared with the CLI, the default conversation Agent) cannot be
* deleted; callers must first drain any active run via manager.abortAgent.
* The directory is deleted recursively (including Trace), and the DB's
* agents/sessions index rows are removed along with it; usage records are kept
* (historical stats are unaffected).
*/
async deleteAgent(projectId: string, agentId: string): Promise<void> {
if (BUILTIN_AGENT_IDS.includes(agentId)) {
throw new HttpError(
409,
"cannot_delete_builtin_agent",
"内置 Agent(default_agent)随 Project 供给,不能从 Web 删除。",
);
}
await fs.rm(agentDir(this.root, projectId, agentId), { recursive: true, force: true });
this.agents.delete(projectId, agentId);
}
/**
* Create an Agent: the id is chosen by the creator (a semantic id, checked for
* duplicates against both the DB and the directory within the Project — a 409
* if taken, which naturally also blocks built-in Agent ids) → initialize State →
* write name/description (name defaults to the id).
*/
async createAgent(
projectId: string,
agentId: string,
name?: string,
description?: string,
): Promise<AgentListItem> {
if (!SEMANTIC_ID_PATTERN.test(agentId)) {
throw new HttpError(400, "invalid_agent_id", `Agent id 须为 2~64 位:${SEMANTIC_ID_RULE}。`);
}
const taken =
this.agents.exists(projectId, agentId) ||
(await fs.stat(agentDir(this.root, projectId, agentId)).then(
() => true,
() => false,
));
if (taken) {
throw new HttpError(409, "agent_exists", `Agent id 已被占用:${agentId}。`);
}
const displayName = name ?? agentId;
await coreCreateAgent({ root: this.root, projectId, agentId });
try {
await this.agentConfig.updateConfig(projectId, agentId, {
config: { name: displayName, ...(description !== undefined ? { description } : {}) },
});
} catch (err) {
// If initialization fails partway through, clean up the directory: an orphaned
// directory would make retries with this agent id 409 forever.
await fs
.rm(agentDir(this.root, projectId, agentId), { recursive: true, force: true })
.catch(() => {});
throw err;
}
const createdAt = new Date().toISOString();
this.agents.insertOrIgnore({ projectId, agentId, createdAt });
// The init template ships with a default toolset and version number; read back the actual values.
const meta = await this.agentConfig.readCardMeta(projectId, agentId);
return {
agentId,
name: displayName,
...(description !== undefined ? { description } : {}),
createdAt,
updatedAt: createdAt,
toolCount: meta.toolCount,
version: meta.version,
vaultKeyCount: 0,
scheduleCount: 0,
};
}
}
@@ -0,0 +1,216 @@
/**
* Benchmark score reading (read-only display): walks `benchmarks/<id>/`, reads
* `benchmark_config.toml` (title,
* description, evaluation Model, per-case run count `runs`) and `scoreboard.yaml`
* (evaluations[], scoreboard v2: each case carries a runs array and a summary).
* Content is created and refined by benchmark_builder; the server only reads it.
* Missing or corrupt files always degrade gracefully (title falls back to the
* directory name, scores come back empty) rather than throwing.
*
* The three per-case metrics trust the file's own values; when missing they're
* computed as the average over the runs array. The old format (no runs at the
* case level, a single session_id) is parsed as a single run — the server backfills
* one run entry.
* Docs: /docs/self-improvement § "Benchmark storage".
*/
import fs from "node:fs/promises";
import path from "node:path";
import { parse as parseToml } from "smol-toml";
import { parse as parseYaml } from "yaml";
import { benchmarksDir } from "@prismshadow/penguin-core";
import type {
BenchmarkCaseScore,
BenchmarkEvaluation,
BenchmarkRunScore,
BenchmarkSummary,
BenchmarksResponse,
} from "../api/types.js";
function asRecord(v: unknown): Record<string, unknown> {
return v !== null && typeof v === "object" && !Array.isArray(v)
? (v as Record<string, unknown>)
: {};
}
function numberOr(v: unknown): number | undefined {
return typeof v === "number" && Number.isFinite(v) ? v : undefined;
}
function stringOr(v: unknown): string | undefined {
return typeof v === "string" && v !== "" ? v : undefined;
}
/** Shapes a single run entry: score is the minimum requirement, other fields tolerate being absent; a bad entry returns null and is dropped. */
function toRun(v: unknown): BenchmarkRunScore | null {
const r = asRecord(v);
const score = numberOr(r.score);
if (score === undefined) return null;
const cost = numberOr(r.cost);
const durationMs = numberOr(r.duration_ms);
const sessionId = stringOr(r.session_id);
return {
score,
...(cost !== undefined ? { cost } : {}),
...(durationMs !== undefined ? { durationMs } : {}),
...(sessionId !== undefined ? { sessionId } : {}),
};
}
/** Average of a metric across runs; undefined when there's no value at all (never forced to 0). */
function averageOf(runs: BenchmarkRunScore[], pick: (r: BenchmarkRunScore) => number | undefined) {
const values = runs.map(pick).filter((v): v is number => v !== undefined);
if (values.length === 0) return undefined;
return values.reduce((a, b) => a + b, 0) / values.length;
}
/**
* Shapes a case-level entry (scoreboard v2): the three metrics trust the file's own
* values, falling back to an average over runs when missing; the old format (no
* runs, a single case-level session_id) is backfilled into a single run. case and a
* score (from the file or derivable from runs) are the minimum requirement,
* otherwise the entry is dropped.
*/
function toCase(v: unknown): BenchmarkCaseScore | null {
const cr = asRecord(v);
const caseId = stringOr(cr.case);
if (caseId === undefined) return null;
const parsedRuns = Array.isArray(cr.runs)
? cr.runs.map(toRun).filter((r): r is BenchmarkRunScore => r !== null)
: [];
const score = numberOr(cr.score) ?? averageOf(parsedRuns, (r) => r.score);
if (score === undefined) return null;
const cost = numberOr(cr.cost) ?? averageOf(parsedRuns, (r) => r.cost);
const durationMs = numberOr(cr.duration_ms) ?? averageOf(parsedRuns, (r) => r.durationMs);
const sessionId = stringOr(cr.session_id);
const runs: BenchmarkRunScore[] =
parsedRuns.length > 0
? parsedRuns
: [
// The old format is parsed as a single run: the case-level values are that run's raw result.
{
score,
...(cost !== undefined ? { cost } : {}),
...(durationMs !== undefined ? { durationMs } : {}),
...(sessionId !== undefined ? { sessionId } : {}),
},
];
return {
case: caseId,
score,
...(cost !== undefined ? { cost } : {}),
...(durationMs !== undefined ? { durationMs } : {}),
...(sessionId !== undefined ? { sessionId } : {}),
runs,
};
}
/** Shapes a single evaluation record: time and score are the minimum requirement, other fields (summary, etc.) tolerate being absent. */
function toEvaluation(v: unknown): BenchmarkEvaluation | null {
const r = asRecord(v);
const time = r.time instanceof Date ? r.time.toISOString() : r.time;
const score = numberOr(r.score);
if (typeof time !== "string" || time === "" || score === undefined) return null;
const cases: BenchmarkCaseScore[] = Array.isArray(r.cases)
? r.cases.map(toCase).filter((c): c is BenchmarkCaseScore => c !== null)
: [];
const summary = stringOr(r.summary);
// Title and body are separate: summary_title is a one-line
// conclusion, summary is the body text.
const summaryTitle = stringOr(r.summary_title);
// The Model actually used for this evaluation run (paired with provider):
// charted curves are split into series by model, each with a distinct color.
const modelId = stringOr(r.model_id);
const provider = stringOr(r.provider);
const version = numberOr(r.version);
const cost = numberOr(r.cost);
const durationMs = numberOr(r.duration_ms);
return {
time,
...(summaryTitle !== undefined ? { summaryTitle } : {}),
...(summary !== undefined ? { summary } : {}),
...(modelId !== undefined ? { modelId } : {}),
...(provider !== undefined ? { provider } : {}),
score,
...(version !== undefined ? { version } : {}),
...(cost !== undefined ? { cost } : {}),
...(durationMs !== undefined ? { durationMs } : {}),
cases,
};
}
export class BenchmarkService {
constructor(private readonly root: string) {}
async list(projectId: string, agentId: string): Promise<BenchmarksResponse> {
const dir = benchmarksDir(this.root, projectId, agentId);
let items: Array<{ name: string; isDir: boolean }>;
try {
const entries = await fs.readdir(dir, { withFileTypes: true });
items = entries.map((e) => ({ name: e.name, isDir: e.isDirectory() }));
} catch {
return { benchmarks: [] }; // Doesn't exist when unconfigured.
}
const benchmarks: BenchmarkSummary[] = [];
for (const item of items.filter((i) => i.isDir).sort((a, b) => a.name.localeCompare(b.name))) {
benchmarks.push(await this.readBenchmark(path.join(dir, item.name), item.name));
}
return { benchmarks };
}
private async readBenchmark(benchDir: string, id: string): Promise<BenchmarkSummary> {
// benchmark_config.toml: title, description, and per-case run count (falls back
// to defaults if corrupt). The model isn't part of the config — each evaluation
// carries the Model actually used for that run.
let title = id;
let description: string | undefined;
let runs: number | undefined;
try {
const config = asRecord(
parseToml(await fs.readFile(path.join(benchDir, "benchmark_config.toml"), "utf8")),
);
if (typeof config.title === "string" && config.title !== "") title = config.title;
if (typeof config.description === "string" && config.description !== "") {
description = config.description;
}
const configRuns = numberOr(config.runs);
if (configRuns !== undefined && Number.isInteger(configRuns) && configRuns >= 1) {
runs = configRuns;
}
} catch {
// Missing or corrupt: title falls back to the directory name.
}
// scoreboard.yaml: evaluations[] is appended over time; bad entries are dropped one by one.
let evaluations: BenchmarkEvaluation[] = [];
try {
const scoreboard = asRecord(
parseYaml(await fs.readFile(path.join(benchDir, "scoreboard.yaml"), "utf8")),
);
if (Array.isArray(scoreboard.evaluations)) {
evaluations = scoreboard.evaluations
.map(toEvaluation)
.filter((e): e is BenchmarkEvaluation => e !== null);
}
} catch {
// No scores yet.
}
// Case count: number of case subfolders (the statement/rubric structure isn't validated here).
let caseCount = 0;
try {
const entries = await fs.readdir(benchDir, { withFileTypes: true });
caseCount = entries.filter((e) => e.isDirectory()).length;
} catch {
// Stays at 0.
}
return {
id,
title,
...(description !== undefined ? { description } : {}),
...(runs !== undefined ? { runs } : {}),
caseCount,
evaluations,
};
}
}
+34
View File
@@ -0,0 +1,34 @@
/**
* id rules: user_id / project_id / agent_id are semantic ids
* chosen by their creator at creation time — starting with a lowercase letter,
* containing only lowercase letters, digits, and underscores. The id doubles as the
* directory name, so keeping it all-lowercase avoids directory name collisions on
* case-insensitive filesystems (e.g. macOS) at the source.
* The hyphen is a **reserved separator**: it only appears at the join point of a
* non-admin project_id's "<username>-<suffix>" concatenation. Usernames never
* contain a hyphen, so the first hyphen is the ownership boundary — no username can
* ever be crafted to collide with another user's prefix, keeping namespaces
* non-overlapping. session_id and temporary workspace ids are still generated by
* the server (randomHex8).
*/
import { randomBytes } from "node:crypto";
/** 8-character lowercase hex random string (used for server-generated ids like session_id / temporary workspace). */
export function randomHex8(): string {
return randomBytes(4).toString("hex");
}
/** General semantic id rule: starts with a lowercase letter, followed by lowercase letters, digits, or underscores only, 2-64 chars (no hyphen). */
export const SEMANTIC_ID_PATTERN = /^[a-z][a-z0-9_]{1,63}$/;
/** Username tightens the general rule to 2-32 chars: leaves headroom for the default Project id `<username>-default_project`. */
export const USERNAME_PATTERN = /^[a-z][a-z0-9_]{1,31}$/;
/** Suffix segment of a non-admin project_id (after `<username>-`): lowercase letters, digits, and underscores only. */
export const PROJECT_SUFFIX_PATTERN = /^[a-z0-9_]+$/;
/** Upper bound on total project_id length (username <=32 + separator + suffix). */
export const PROJECT_ID_MAX_LENGTH = 64;
/** Human-readable description of the rule (reused in error messages). */
export const SEMANTIC_ID_RULE = "小写字母开头,仅小写字母、数字与下划线";
@@ -0,0 +1,498 @@
/**
* `.project_config.toml` read/write (single hidden config file).
*
* Doesn't reuse core's loadProjectConfig/saveProjectConfig (they only keep known
* fields): reads and writes the complete object directly via smol-toml, preserving
* extension fields like `name`. credential (api_key / base_url / created_at) is
* **inlined on the model entry** — there's no longer a supplementary section or
* secrets file; since the file contains secrets, it's always written with mode
* 0600. Plaintext only ever hits disk, and is always masked in responses.
*
* Model references are **fully split into separate fields**: an entry is
* stored as two independent fields, `provider` and `model_id`; the `(provider,
* model_id)` pair is the entry's unique key. `model_id` is the upstream request id,
* sent to AgentHub verbatim — string concatenation like `<provider>/<id>` is
* forbidden everywhere in the pipeline. `default_model` / `vision_model` are `{
* provider, model_id }` paired references (TOML tables).
*/
import fs from "node:fs/promises";
import path from "node:path";
import { parse as parseToml } from "smol-toml";
import {
GenerativeModel,
catalogEntryFor,
defaultProjectConfig,
projectConfigPath,
renderProjectConfigToml,
resolveModelEnv,
userText,
} from "@prismshadow/penguin-core";
import type { ModelRef } from "@prismshadow/penguin-core";
import type {
ModelInfo,
ModelPricingDto,
ModelRefDto,
ModelsResponse,
ModelsUpdateRequest,
ModelTestRequest,
ModelTestResponse,
} from "../api/types.js";
import { badRequest } from "../http/validate.js";
import type { PricingRates } from "./usage-service.js";
type RawTable = Record<string, unknown>;
/**
* API key masking: length <=12 -> `***`, otherwise `first4…last4`; plaintext is
* never sent to the client. The 12-char threshold: `first4…last4` exposes 8
* characters, which for a 9-12 character short secret would leak more than half of
* it, so those are masked in full instead.
*/
export function maskApiKey(key: string): string {
if (key.length <= 12) return "***";
return `${key.slice(0, 4)}…${key.slice(-4)}`;
}
function asTable(v: unknown): RawTable {
return v !== null && typeof v === "object" && !Array.isArray(v) ? (v as RawTable) : {};
}
function asArray(v: unknown): RawTable[] {
return Array.isArray(v) ? v.map(asTable) : [];
}
function optNum(v: unknown): number | undefined {
return typeof v === "number" && Number.isFinite(v) ? v : undefined;
}
function optStr(v: unknown): string | undefined {
return typeof v === "string" && v !== "" ? v : undefined;
}
/** Leniently reads a paired reference table (default_model / vision_model); returns undefined on a shape mismatch (including the old string format). */
function optRef(v: unknown): ModelRef | undefined {
const t = asTable(v);
const provider = optStr(t.provider);
const modelId = optStr(t.model_id);
return provider !== undefined && modelId !== undefined
? { provider, model_id: modelId }
: undefined;
}
/** Whether an entry matches a paired reference (the entry's provider / model_id fields must be strings). */
function entryMatches(m: RawTable, provider: string, modelId: string): boolean {
return m.provider === provider && m.model_id === modelId;
}
/** In-process Map/Set key for a paired reference (\0-separated to avoid concatenation ambiguity; never persisted, not an id format). */
function refKey(provider: string, modelId: string): string {
return `${provider}\0${modelId}`;
}
/** Display form of a paired reference (for error messages; display only, not a storage format). */
function showRef(provider: string, modelId: string): string {
return `(provider=${provider}, model_id=${modelId})`;
}
export class ProjectConfigService {
constructor(private readonly root: string) {}
private filePath(projectId: string): string {
return projectConfigPath(this.root, projectId);
}
/** Reads the raw TOML object; returns an empty object if the file doesn't exist (does not write to disk). */
async readRaw(projectId: string): Promise<RawTable> {
let raw: string;
try {
raw = await fs.readFile(this.filePath(projectId), "utf8");
} catch (err) {
if ((err as NodeJS.ErrnoException).code === "ENOENT") return {};
throw err;
}
return asTable(parseToml(raw));
}
/**
* Writes the whole object to disk: the file inlines secrets like api_key, always
* written with mode 0600 (the `mode` option only applies at creation time, so
* chmod is used to enforce it on existing files too — matching core's
* saveProjectConfig behavior).
*/
async writeRaw(projectId: string, data: RawTable): Promise<void> {
const file = this.filePath(projectId);
await fs.mkdir(path.dirname(file), { recursive: true });
// Rendering goes through core's single writer: paired references become inline
// tables, models is placed last — matching the CLI's output format exactly
// (the same file should never have two formats).
await fs.writeFile(file, renderProjectConfigToml(data), { encoding: "utf8", mode: 0o600 });
await fs.chmod(file, 0o600);
}
/**
* Initial config for a newly created Project: display name + preset built-in
* model catalog (the default model and all preset entries, sourced from the same
* core defaultProjectConfig; a gateway model's base_url is already inlined on the
* entry, with no key); users only need to fill in an API key as needed (leave it
* blank to fall back to the provider's environment variable).
*/
async writeInitialConfig(projectId: string, name: string): Promise<void> {
const preset = defaultProjectConfig();
await this.writeRaw(projectId, {
name,
...(preset.default_model !== undefined ? { default_model: preset.default_model } : {}),
models: preset.models,
});
}
/**
* Backfills preset models (for onboarding an existing Project, e.g. the
* `default_project` shared with the CLI when the first user is onboarded — its
* directory already existed and never went through `writeInitialConfig`, so it
* previously had no models and no default model).
*
* **Only backfills when there are no models at all**: a Project that already has
* models configured (via the CLI or edited by the user) is left as-is, and its
* other fields (name, etc.) are preserved too — existing config is never
* overwritten.
*/
async ensurePresetModels(projectId: string): Promise<void> {
const raw = await this.readRaw(projectId);
if (asArray(raw.models).length > 0) return;
const preset = defaultProjectConfig();
await this.writeRaw(projectId, {
...raw,
// Also reset to the preset default_model if the existing one points at a now-deleted model, to keep the default model valid.
...(preset.default_model !== undefined ? { default_model: preset.default_model } : {}),
models: preset.models,
});
}
/** Project display name (the toml's name; returns undefined if unset, the frontend falls back to displaying the id). */
async getName(projectId: string): Promise<string | undefined> {
const raw = await this.readRaw(projectId);
return typeof raw.name === "string" ? raw.name : undefined;
}
/** Paired reference of the default Model; returns undefined if unconfigured (or in the old string format). */
async getDefaultModelRef(projectId: string): Promise<ModelRef | undefined> {
const raw = await this.readRaw(projectId);
return optRef(raw.default_model);
}
/** Pricing lookup for usage-recorder: the current pricing for this paired reference (undefined if none -> cost is NULL). */
async getPricing(
projectId: string,
provider: string,
modelId: string,
): Promise<PricingRates | undefined> {
const raw = await this.readRaw(projectId);
const entry = asArray(raw.models).find((m) => entryMatches(m, provider, modelId));
const pricing = entry ? asTable(entry.pricing) : {};
const cacheRead = optNum(pricing.cache_read);
const cacheWrite = optNum(pricing.cache_write);
const output = optNum(pricing.output);
if (cacheRead === undefined && cacheWrite === undefined && output === undefined) {
return undefined;
}
return { cacheRead: cacheRead ?? 0, cacheWrite: cacheWrite ?? 0, output: output ?? 0 };
}
/**
* Model connectivity test: the model reference `(provider, modelId)` is submitted
* as a pair in the request body; sends one minimal request using that model's
* config (optionally overridden with an unsaved apiKey / baseUrl) — no tools, no
* system prompt, thinking disabled, a tiny output cap, 20s timeout — just to see
* whether it completes normally. The model id sent to AgentHub is `modelId`
* itself (the upstream id verbatim; client_type inference follows it).
*
* Never throws: the LLM layer collapses auth/parameter/network errors into an
* `LLMOutcome`, which is translated here into ok / message. Consumes very few
* Tokens (single-digit output), and writes no Trace and records no usage.
*/
async testModel(projectId: string, req: ModelTestRequest): Promise<ModelTestResponse> {
const raw = await this.readRaw(projectId);
// Testable even if the model isn't in the config yet (validate before saving when adding a custom model): in that case all parameters come from the request body.
const entry = asArray(raw.models).find((m) => entryMatches(m, req.provider, req.modelId)) ?? {};
// Always tests against the **current form draft**: checking "clear" means the saved key is not fallen back to; an explicit null base URL is treated as cleared.
const savedKey = optStr(entry.api_key);
const apiKey = req.clearApiKey ? undefined : (req.apiKey ?? savedKey);
const savedBaseUrl = optStr(entry.base_url);
const baseUrl = req.baseUrl === null ? undefined : (req.baseUrl ?? savedBaseUrl);
const clientType = req.clientType ?? optStr(entry.client_type);
const startedAt = Date.now();
try {
// Construction must be inside the try block: the underlying provider SDK can
// throw during **client construction** itself when a credential is missing
// (models on the OpenAI protocol need apiKey/OPENAI_API_KEY) — the whole point
// of a connectivity test is to collapse that kind of failure into
// `{ ok:false }`; if construction were outside the try, a missing-key test
// would bubble up as a 500.
const llm = new GenerativeModel({
modelId: req.modelId,
...(apiKey ? { apiKey } : {}),
...(baseUrl ? { baseUrl } : {}),
...(clientType ? { clientType } : {}),
tools: [],
thinkingLevel: "none",
maxTokens: 16,
requestTimeoutMs: 20_000,
});
const gen = llm.streamGenerate({ newMessages: [userText("ping")] });
for (;;) {
const step = await gen.next();
if (step.done) {
const outcome = step.value;
if (outcome.status === "completed")
return { ok: true, latencyMs: Date.now() - startedAt };
const detail = "message" in outcome && outcome.message ? outcome.message : outcome.status;
return { ok: false, message: String(detail).slice(0, 300) };
}
}
} catch (err) {
// Defensive: an unexpected exception during construction/iteration (the LLM layer promises not to throw; this is a fallback).
return {
ok: false,
message: (err instanceof Error ? err.message : String(err)).slice(0, 300),
};
}
}
/**
* GET models view: masks credential (inline fields), flags the default Model;
* the group is the entry's `provider` field, looked up in the built-in catalog by
* the `(provider, model_id)` pair to fill in displayName / envKey (entries outside
* the catalog are treated as custom models: envKey only has a fallback for the
* openai protocol). vision follows the TOML annotation when present, otherwise
* falls back to the catalog annotation (if neither exists, the field is omitted =
* supported by default).
*/
async getModels(projectId: string): Promise<ModelsResponse> {
const raw = await this.readRaw(projectId);
const defaultRef = optRef(raw.default_model);
const visionRef = optRef(raw.vision_model);
const models: ModelInfo[] = asArray(raw.models)
// An entry is valid only if both provider and model_id are strings (an entry in the old concatenated format lacks provider and is ignored).
.filter((m) => typeof m.provider === "string" && typeof m.model_id === "string")
.map((m) => {
const provider = m.provider as string;
const modelId = m.model_id as string;
const pricing = asTable(m.pricing);
const pricingDto: ModelPricingDto | undefined =
optNum(pricing.cache_read) !== undefined ||
optNum(pricing.cache_write) !== undefined ||
optNum(pricing.output) !== undefined
? {
cacheRead: optNum(pricing.cache_read) ?? 0,
cacheWrite: optNum(pricing.cache_write) ?? 0,
output: optNum(pricing.output) ?? 0,
}
: undefined;
const clientType = optStr(m.client_type);
const cat = catalogEntryFor(provider, modelId);
// The env fallback is reported as-is: follows the same rule as
// AgentHub routing — an explicit client_type takes priority (the openai
// protocol reads OPENAI_*, independent of the group), otherwise it's
// auto-routed to a provider client based on model_id; an id that can't be
// routed has no fallback (no envKey, and AgentHub will reject that id).
const envKey = resolveModelEnv(modelId, clientType)?.envKey;
const vision = typeof m.vision === "boolean" ? m.vision : cat?.supportsVision;
// Display name: the explicit TOML field (user-edited) takes priority, then the built-in catalog.
const displayName = optStr(m.display_name) ?? cat?.displayName;
// credential is inlined on the entry: a credential block is emitted if either api_key or base_url is present.
const apiKey = optStr(m.api_key);
const credBaseUrl = optStr(m.base_url);
const createdAt = optStr(m.created_at);
const info: ModelInfo = {
provider,
modelId,
...(displayName !== undefined ? { displayName } : {}),
isDefault:
defaultRef !== undefined &&
defaultRef.provider === provider &&
defaultRef.model_id === modelId,
...(optNum(m.context_window) !== undefined
? { contextWindow: optNum(m.context_window)! }
: {}),
...(clientType ? { clientType } : {}),
...(vision !== undefined ? { vision } : {}),
...(envKey ? { envKey } : {}),
...(pricingDto ? { pricing: pricingDto } : {}),
...(apiKey !== undefined || credBaseUrl !== undefined
? {
credential: {
...(apiKey !== undefined ? { apiKeyMasked: maskApiKey(apiKey) } : {}),
...(credBaseUrl !== undefined ? { baseUrl: credBaseUrl } : {}),
...(createdAt !== undefined ? { createdAt } : {}),
},
}
: {}),
};
return info;
});
const toDto = (ref: ModelRef): ModelRefDto => ({
provider: ref.provider,
modelId: ref.model_id,
});
return {
...(defaultRef !== undefined ? { defaultModel: toDto(defaultRef) } : {}),
...(visionRef !== undefined ? { visionModel: toDto(visionRef) } : {}),
models,
};
}
/**
* PUT replaces the whole models table: key =
* `(provider, modelId)`; model entries that no longer appear are deleted along
* with their inline credential; omitting apiKey keeps the existing value,
* providing one overwrites it and records created_at, clearApiKey clears it;
* baseUrl null clears it / omitted keeps it. A key change (either the group or
* the upstream id changes) is migrated as a pair via `renamedFrom`: credential and
* unknown fields migrate along with the base entry, and default/vision pointers
* follow. Other extension fields in the toml (name, etc.) are preserved.
*/
async updateModels(projectId: string, req: ModelsUpdateRequest): Promise<ModelsResponse> {
const raw = await this.readRaw(projectId);
const prevModels = asArray(raw.models);
const seen = new Set<string>();
const nextModels: RawTable[] = [];
// Rename mapping (old reference key -> new reference): default model / vision model pointers follow a key change instead of being lost on a full table replacement.
const renamed = new Map<string, ModelRefDto>();
for (const entry of req.models) {
const key = refKey(entry.provider, entry.modelId);
if (seen.has(key)) {
throw badRequest(
`models 中存在重复的模型引用:${showRef(entry.provider, entry.modelId)}。`,
);
}
seen.add(key);
if (
entry.renamedFrom !== undefined &&
!(
entry.renamedFrom.provider === entry.provider &&
entry.renamedFrom.modelId === entry.modelId
)
) {
renamed.set(refKey(entry.renamedFrom.provider, entry.renamedFrom.modelId), {
provider: entry.provider,
modelId: entry.modelId,
});
}
// Model entry: uses the old entry (the entry for the original reference when
// the key changed) as the base, preserving unknown fields and inline
// credential; known fields are replaced wholesale per the request (omitted
// means removed).
const prevRef = entry.renamedFrom ?? { provider: entry.provider, modelId: entry.modelId };
const prev = prevModels.find((m) => entryMatches(m, prevRef.provider, prevRef.modelId)) ?? {};
const next: RawTable = { ...prev, provider: entry.provider, model_id: entry.modelId };
delete next.context_window;
delete next.client_type;
delete next.vision;
delete next.pricing;
delete next.display_name;
// Leftover key from the old concatenated format (request_model_id): defensively stripped, never written to disk again.
delete next.request_model_id;
// Display name: **only written to disk when it differs from the built-in
// catalog (looked up by the paired reference)** — preset models keep the
// config clean, only user-edited ones (including those not found in the
// catalog) get written into the TOML.
const catNew = catalogEntryFor(entry.provider, entry.modelId);
if (entry.displayName && entry.displayName !== catNew?.displayName) {
next.display_name = entry.displayName;
}
if (entry.contextWindow !== undefined) next.context_window = entry.contextWindow;
if (entry.clientType) next.client_type = entry.clientType;
// Treated as supported by default: only written to disk when explicitly annotated (both true/false are kept; false drives a frontend blocking hint).
if (entry.vision !== undefined) next.vision = entry.vision;
if (entry.pricing !== undefined) {
next.pricing = {
unit: "usd_per_mtok",
cache_read: entry.pricing.cacheRead,
cache_write: entry.pricing.cacheWrite,
output: entry.pricing.output,
};
}
// credential is inlined on the entry; added/removed on top of the old value per the request (migrates automatically with the base entry when the key changes).
if (entry.clearApiKey) {
delete next.api_key;
delete next.created_at;
}
if (entry.apiKey !== undefined) {
next.api_key = entry.apiKey;
next.created_at = new Date().toISOString();
}
if (entry.baseUrl === null) delete next.base_url;
else if (entry.baseUrl !== undefined) next.base_url = entry.baseUrl;
nextModels.push(next);
}
// default_model: when provided it must be present in models; when omitted the previous value is kept (the pointer follows a key rename; if it was deleted, it's removed).
let defaultModel: ModelRefDto | undefined;
if (req.defaultModel !== undefined) {
if (!seen.has(refKey(req.defaultModel.provider, req.defaultModel.modelId))) {
throw badRequest(
`defaultModel 必须包含在 models 内:${showRef(req.defaultModel.provider, req.defaultModel.modelId)}。`,
);
}
defaultModel = req.defaultModel;
} else {
const prevRef = optRef(raw.default_model);
if (prevRef !== undefined) {
const prevKey = refKey(prevRef.provider, prevRef.model_id);
const followed = renamed.get(prevKey) ?? {
provider: prevRef.provider,
modelId: prevRef.model_id,
};
if (seen.has(refKey(followed.provider, followed.modelId))) defaultModel = followed;
}
}
// vision_model: same semantics as default_model; additionally must not be annotated vision=false (can't proxy-read images if unsupported).
const targetOf = (ref: ModelRefDto) =>
req.models.find((m) => m.provider === ref.provider && m.modelId === ref.modelId);
let visionModel: ModelRefDto | undefined;
if (req.visionModel !== undefined) {
if (!seen.has(refKey(req.visionModel.provider, req.visionModel.modelId))) {
throw badRequest(
`visionModel 必须包含在 models 内:${showRef(req.visionModel.provider, req.visionModel.modelId)}。`,
);
}
if (targetOf(req.visionModel)?.vision === false) {
throw badRequest(
`visionModel 不能指向标注为不支持图片的模型:${showRef(req.visionModel.provider, req.visionModel.modelId)}。`,
);
}
visionModel = req.visionModel;
} else {
const prevRef = optRef(raw.vision_model);
if (prevRef !== undefined) {
const prevKey = refKey(prevRef.provider, prevRef.model_id);
const followed = renamed.get(prevKey) ?? {
provider: prevRef.provider,
modelId: prevRef.model_id,
};
if (seen.has(refKey(followed.provider, followed.modelId))) {
// The former vision model is now annotated as not supporting images: the annotation takes priority, and the pointer is dropped as invalid.
if (targetOf(followed)?.vision !== false) visionModel = followed;
}
}
}
const toRaw = (ref: ModelRefDto): RawTable => ({
provider: ref.provider,
model_id: ref.modelId,
});
const next: RawTable = { ...raw, models: nextModels };
if (defaultModel !== undefined) next.default_model = toRaw(defaultModel);
else delete next.default_model;
if (visionModel !== undefined) next.vision_model = toRaw(visionModel);
else delete next.vision_model;
await this.writeRaw(projectId, next);
return this.getModels(projectId);
}
}
@@ -0,0 +1,350 @@
/**
* Project service.
*
* The single implementation point for authorization rules: `requireProjectAccess`
* (owner or member, otherwise 404 without leaking existence) and
* `requireProjectOwner` (owner only; 403 when known to be accessible, 404 when not)
* are reused by every route; the non-throwing `canAccess` (for error attribution)
* is likewise just a sibling wrapper around them — all three share the single
* `resolveAccess` decision, with no second rule set maintained separately.
* Also handles Project create / list / delete, member authorization, and initial
* Project provisioning at signup.
*/
import fs from "node:fs/promises";
import { DEFAULT_PROJECT_ID, projectDir, provisionProjectAgents } from "@prismshadow/penguin-core";
import type { MemberInfo, ProjectRole, ProjectSummary } from "../api/types.js";
import { HttpError } from "../http/errors.js";
import type { AgentsRepo } from "../db/repos/agents.js";
import type { ErrorsRepo } from "../db/repos/errors.js";
import type { MembersRepo } from "../db/repos/members.js";
import type { ProjectRow, ProjectsRepo } from "../db/repos/projects.js";
import type { SessionsRepo } from "../db/repos/sessions.js";
import type { SchedulesRepo } from "../db/repos/schedules.js";
import type { UsageRepo } from "../db/repos/usage.js";
import type { UserRow, UsersRepo } from "../db/repos/users.js";
import type { SessionManager } from "../runtime/session-manager.js";
import {
PROJECT_ID_MAX_LENGTH,
PROJECT_SUFFIX_PATTERN,
SEMANTIC_ID_PATTERN,
SEMANTIC_ID_RULE,
} from "./ids.js";
import type { ProjectConfigService } from "./project-config-service.js";
/** Fallback timeout for waiting on runs to settle before deleting a Project. */
const ABORT_SETTLE_TIMEOUT_MS = 5000;
async function dirExists(path: string): Promise<boolean> {
try {
const stat = await fs.stat(path);
return stat.isDirectory();
} catch {
return false;
}
}
export interface ProjectServiceDeps {
root: string;
users: UsersRepo;
projects: ProjectsRepo;
members: MembersRepo;
agents: AgentsRepo;
sessions: SessionsRepo;
usage: UsageRepo;
errors: ErrorsRepo;
schedules: SchedulesRepo;
projectConfig: ProjectConfigService;
manager: SessionManager;
}
export class ProjectService {
constructor(private readonly deps: ProjectServiceDeps) {}
// —— Authorization rules (single implementation point) ——
/**
* The **sole** implementation of the owner / member check: returns the row with
* a role if accessible, otherwise null. `requireProjectAccess` below (throws 404)
* and `canAccess` (returns boolean) are both just wrappers around it — there's
* only one copy of the decision rule, since writing it twice would eventually
* drift out of sync.
*/
private resolveAccess(
userId: string,
projectId: string,
): (ProjectRow & { role: ProjectRole }) | null {
const row = this.deps.projects.findById(projectId);
if (!row) return null;
if (row.ownerUserId === userId) return { ...row, role: "owner" };
if (this.deps.members.isMember(projectId, userId)) return { ...row, role: "member" };
return null;
}
/** Accessible by owner or member; otherwise 404 (does not leak Project existence). */
requireProjectAccess(userId: string, projectId: string): ProjectRow & { role: ProjectRole } {
const row = this.resolveAccess(userId, projectId);
if (!row) {
throw new HttpError(404, "project_not_found", "Project 不存在或无权访问。");
}
return row;
}
/**
* The non-throwing version of the same check: used for **error attribution**
* (app.onError) — that runs on the error-handling path, where throwing another
* 404 would only break error handling; whether access is granted shouldn't be
* expressed as an exception there anyway.
*/
canAccess(userId: string, projectId: string): boolean {
return this.resolveAccess(userId, projectId) !== null;
}
/** Owner only: 403 when known accessible as a member; 404 when not accessible. */
requireProjectOwner(userId: string, projectId: string): ProjectRow {
const row = this.requireProjectAccess(userId, projectId);
if (row.role !== "owner") {
throw new HttpError(403, "owner_required", "该操作仅 Project owner 可执行。");
}
return row;
}
/** List of project_ids accessible to the current user (owned + granted access) (used by workspace-guard). */
accessibleProjectIds(userId: string): string[] {
return this.deps.projects.listAccessible(userId).map((p) => p.projectId);
}
// —— Project lifecycle ——
/** List of owned + granted-access Projects; display names are read from each project_config.toml. */
async listProjects(userId: string): Promise<ProjectSummary[]> {
const rows = this.deps.projects.listAccessible(userId);
return Promise.all(
rows.map(async (row) => {
const name = await this.deps.projectConfig.getName(row.projectId);
return {
projectId: row.projectId,
...(name !== undefined ? { name } : {}),
role: row.role,
ownerUserId: row.ownerUserId,
createdAt: row.createdAt,
};
}),
);
}
/**
* Create a Project: the id is chosen by the creator (a semantic id, checked for
* duplicates against both the DB and the directory — 409 if taken), the initial
* config is written (display name defaults to the id), and the built-in Agent is
* initialized.
* A non-admin's id is forced to be "<username>-<suffix>", where the suffix is
* lowercase letters, digits, and underscores only — the hyphen is a reserved
* separator, usernames never contain a hyphen, so the first hyphen is the
* ownership boundary and the prefix can never be crafted from another username;
* an admin's id contains no hyphen (occupying no user's namespace).
*/
async createProject(owner: UserRow, projectId: string, name?: string): Promise<ProjectSummary> {
if (owner.isAdmin) {
if (!SEMANTIC_ID_PATTERN.test(projectId)) {
throw new HttpError(
400,
"invalid_project_id",
`Project id 须为 2~64 位:${SEMANTIC_ID_RULE}(连字符保留作用户命名空间分隔)。`,
);
}
} else {
const prefix = `${owner.userId}-`;
const suffix = projectId.startsWith(prefix) ? projectId.slice(prefix.length) : "";
if (!PROJECT_SUFFIX_PATTERN.test(suffix) || projectId.length > PROJECT_ID_MAX_LENGTH) {
throw new HttpError(
400,
"project_id_prefix_required",
`Project id 须以 ${prefix} 开头,后接小写字母、数字或下划线(总长不超过 ${PROJECT_ID_MAX_LENGTH})。`,
);
}
}
if (
this.deps.projects.findById(projectId) !== null ||
(await dirExists(projectDir(this.deps.root, projectId)))
) {
throw new HttpError(409, "project_exists", `Project id 已被占用:${projectId}。`);
}
const displayName = name ?? projectId;
const createdAt = new Date().toISOString();
// Insert the DB row first: the primary key is the final arbiter for concurrent
// creation with the same id (the duplicate check above has an await gap), and a
// conflict is mapped to 409 with **no cleanup** — the directory belongs to the
// winner, and cleaning up here would wrongly delete the other side's data.
try {
this.deps.projects.insert({ projectId, ownerUserId: owner.userId, createdAt });
} catch (err) {
if (err instanceof Error && err.message.includes("UNIQUE")) {
throw new HttpError(409, "project_exists", `Project id 已被占用:${projectId}。`);
}
throw err;
}
// If file initialization fails, roll back the DB row and clean up the
// directory: an orphaned directory would make retries with this id 409 forever
// (a typical scenario: signup failure rolled back the user row, but the
// <username>-default_project directory was left behind).
try {
await fs.mkdir(projectDir(this.deps.root, projectId), { recursive: true });
await this.deps.projectConfig.writeInitialConfig(projectId, displayName);
await this.provisionBuiltinAgents(projectId);
} catch (err) {
this.deps.projects.delete(projectId);
await fs
.rm(projectDir(this.deps.root, projectId), { recursive: true, force: true })
.catch(() => {});
throw err;
}
return {
projectId,
name: displayName,
role: "owner",
ownerUserId: owner.userId,
createdAt,
};
}
/**
* Initial Project provisioned at signup:
* the built-in admin adopts `default_project` (if the directory already exists,
* it's adopted directly without overwriting existing config — shared with the
* CLI); other users get `<username>-default_project` created, with display name
* defaulting to the username.
*/
async provisionInitialProject(user: UserRow, isAdmin: boolean): Promise<void> {
if (!isAdmin) {
await this.createProject(user, `${user.userId}-${DEFAULT_PROJECT_ID}`, user.userId);
return;
}
const projectId = DEFAULT_PROJECT_ID;
// Initialize the built-in Agent (loaded without overwriting if it already exists); this also ensures the directory exists.
await this.provisionBuiltinAgents(projectId);
// Adopting an existing directory doesn't go through writeInitialConfig: preset
// models and the default model are backfilled instead (only when there are no
// models at all; a default_project already configured via the CLI is left
// as-is).
await this.deps.projectConfig.ensurePresetModels(projectId);
this.deps.projects.insert({
projectId,
ownerUserId: user.userId,
createdAt: new Date().toISOString(),
});
}
/**
* Delete a Project (owner): default_project is refused; deleting the user's
* **last accessible Project** is refused too (deleting it would leave the list
* empty, with no Project to select in the Web client and the page stuck on a
* skeleton screen — a typical case being a non-first user deleting the initial
* Project provisioned at signup); active runs are drained first, then the DB and
* directory are cleared.
*/
async deleteProject(userId: string, projectId: string): Promise<void> {
this.requireProjectOwner(userId, projectId);
if (projectId === DEFAULT_PROJECT_ID) {
throw new HttpError(
409,
"cannot_delete_default_project",
"default_project 与 CLI 共用,不能从 Web 删除。",
);
}
if (this.deps.projects.listAccessible(userId).length <= 1) {
throw new HttpError(
409,
"cannot_delete_last_project",
"这是当前账号最后一个 Project,删除后将无 Project 可用;请先创建新的 Project。",
);
}
await this.destroyProject(projectId);
}
/**
* The actual deletion (no authorization or protection checks): shared by
* deleteProject and the cascade cleanup when an admin deletes a user.
* Abort follow-up (writing the abort event to Trace, etc.) happens
* asynchronously: waits for runs to settle (capped at 5s) before deleting the
* directory, to avoid the Trace writer recreating the directory after deletion.
*/
async destroyProject(projectId: string): Promise<void> {
const runnings = this.deps.manager.abortProject(projectId);
if (runnings.length > 0) {
await Promise.race([
Promise.allSettled(runnings).then(() => undefined),
new Promise<void>((resolve) => setTimeout(resolve, ABORT_SETTLE_TIMEOUT_MS).unref?.()),
]);
}
this.deps.projects.delete(projectId); // project_members cascade-deleted
this.deps.agents.deleteByProject(projectId);
this.deps.sessions.deleteByProject(projectId);
this.deps.usage.deleteByProject(projectId);
this.deps.errors.deleteByProject(projectId);
this.deps.schedules.deleteByProject(projectId);
await fs.rm(projectDir(this.deps.root, projectId), { recursive: true, force: true });
}
// —— Member authorization ——
/** Member list: owner (role=owner) + members. */
listMembers(userId: string, projectId: string): MemberInfo[] {
const project = this.requireProjectAccess(userId, projectId);
const members = this.deps.members.list(projectId);
return [
{ userId: project.ownerUserId, role: "owner", createdAt: project.createdAt },
...members.map((m) => ({
userId: m.userId,
role: "member" as const,
createdAt: m.createdAt,
})),
];
}
/** Grant member access (owner): invites by username; 404 if the user doesn't exist, 409 for the owner themself or an existing member. */
addMember(userId: string, projectId: string, targetUserId: string): MemberInfo {
const project = this.requireProjectOwner(userId, projectId);
const target = this.deps.users.findById(targetUserId);
if (!target) {
throw new HttpError(404, "user_not_found", `用户不存在:${targetUserId}。`);
}
if (target.userId === project.ownerUserId) {
throw new HttpError(409, "already_owner", "owner 无需授权给自己。");
}
if (this.deps.members.isMember(projectId, target.userId)) {
throw new HttpError(409, "already_member", `${targetUserId} 已是该 Project 的成员。`);
}
const createdAt = new Date().toISOString();
this.deps.members.insert({ projectId, userId: target.userId, createdAt });
return { userId: target.userId, role: "member", createdAt };
}
/** Revoke member access (owner). */
removeMember(userId: string, projectId: string, targetUserId: string): void {
this.requireProjectOwner(userId, projectId);
if (!this.deps.members.isMember(projectId, targetUserId)) {
throw new HttpError(404, "member_not_found", `该 Project 没有成员:${targetUserId}。`);
}
this.deps.members.delete(projectId, targetUserId);
}
/**
* Ensures the Project's built-in Agent exists (the sole built-in Agent
* default_agent; initialized if the directory is empty, otherwise loaded without
* overwriting) and indexes it. createdAt increments by 1ms in preset order, so
* built-in Agents stably sort first; other Agents backfilled by directory
* scanning are sorted by their own createdAt and are outside the scope of this
* guarantee.
*/
private async provisionBuiltinAgents(projectId: string): Promise<void> {
const agentIds = await provisionProjectAgents({ root: this.deps.root, projectId });
const base = Date.now();
agentIds.forEach((agentId, i) => {
this.deps.agents.insertOrIgnore({
projectId,
agentId,
createdAt: new Date(base + i).toISOString(),
});
});
}
}
@@ -0,0 +1,302 @@
/**
* Session index service.
*
* The list is DB index ∪ Trace directory discovery: scans
* `<agent>/traces/<date>/<session_id>_<index3>.jsonl`; an unmanaged Session (e.g.
* one started via the CLI) has its first line's session_meta read for
* (provider, model_id) / workspace, which is backfilled into a DB row
* (approval_mode defaults, createdAt is taken from the timestamp embedded in
* session_id).
* Create: via core's `agent.createSession` (model reference as a provider + modelId
* pair; defaults to the Project's default reference, 400 if there is none; omitting
* provider goes through resolveModelRef for unique resolution); the new Session is
* added to session-manager's active table (state idle).
*/
import path from "node:path";
import { readdir } from "node:fs/promises";
import {
createAgent,
isSessionMeta,
readTraceTolerant,
tracesDir,
} from "@prismshadow/penguin-core";
import type { ApprovalMode, SessionInfo } from "../api/types.js";
import { HttpError, isMissingCredential, modelCredentialMissing } from "../http/errors.js";
import type { SessionRow, SessionsRepo } from "../db/repos/sessions.js";
import type { SessionManager } from "../runtime/session-manager.js";
import type { ProjectConfigService } from "./project-config-service.js";
const TRACE_FILE_RE = /^(.+)_(\d{3})\.jsonl$/;
const SESSION_ID_TS_RE = /^session-(\d{4})-(\d{2})-(\d{2})-(\d{2})-(\d{2})-(\d{2})-[0-9a-f]{8}$/;
/** Derives creation time from the local timestamp embedded in session_id; returns null if it doesn't match. */
export function sessionIdCreatedAt(sessionId: string): string | null {
const m = SESSION_ID_TS_RE.exec(sessionId);
if (!m) return null;
const [, y, mo, d, h, mi, s] = m;
const date = new Date(Number(y), Number(mo) - 1, Number(d), Number(h), Number(mi), Number(s));
return Number.isNaN(date.getTime()) ? null : date.toISOString();
}
export interface SessionServiceDeps {
root: string;
sessions: SessionsRepo;
manager: SessionManager;
projectConfig: ProjectConfigService;
}
export class SessionService {
constructor(private readonly deps: SessionServiceDeps) {}
/** DB row -> SessionInfo (run status and pending approval count come from session-manager). */
toInfo(row: SessionRow, hasTrace: boolean): SessionInfo {
return {
sessionId: row.sessionId,
projectId: row.projectId,
agentId: row.agentId,
provider: row.provider,
modelId: row.modelId,
workspace: row.workspace,
approvalMode: row.approvalMode,
...(row.title !== null ? { title: row.title } : {}),
...(row.source != null ? { source: row.source } : {}),
createdAt: row.createdAt,
status: this.deps.manager.statusOf(row.sessionId),
pendingApprovalCount: this.deps.manager.pendingApprovalCount(row.sessionId),
hasTrace,
archived: (row.archivedAt ?? null) !== null,
};
}
/** Whether this Session already has a Trace record (a Task has been run). */
async hasTrace(row: SessionRow): Promise<boolean> {
const ids = await this.discoverTraceSessionIds(row.projectId, row.agentId);
return ids.has(row.sessionId);
}
/** List: DB ∪ Trace directory discovery, sorted by createdAt descending. */
async listSessions(projectId: string, agentId: string): Promise<SessionInfo[]> {
const traceIds = await this.discoverTraceSessionIds(projectId, agentId);
const rows = new Map(
this.deps.sessions.listByAgent(projectId, agentId).map((r) => [r.sessionId, r]),
);
// Unmanaged Trace Sessions: backfill an index row by reading the first line's session_meta.
for (const sessionId of traceIds) {
if (rows.has(sessionId)) continue;
const discovered = await this.adoptTraceSession(projectId, agentId, sessionId);
if (discovered) rows.set(sessionId, discovered);
}
return [...rows.values()]
.sort(
(a, b) => b.createdAt.localeCompare(a.createdAt) || b.sessionId.localeCompare(a.sessionId),
)
.map((row) => this.toInfo(row, traceIds.has(row.sessionId)));
}
/**
* Session stats (Agents list card): total count = size of the union of DB index
* ∪ Trace directory discovery; activity = number of active Sessions per day over
* the last `days` days (deduplicated count of Sessions created that day or with a
* Trace record that day; index 0 = earliest, last index = today). Counts only —
* does not backfill index rows.
*/
async sessionStats(
projectId: string,
agentId: string,
days: number,
): Promise<{ sessionCount: number; activity: number[] }> {
const all = new Set<string>();
const byDate = new Map<string, Set<string>>();
const mark = (date: string, sessionId: string): void => {
all.add(sessionId);
const set = byDate.get(date) ?? new Set<string>();
set.add(sessionId);
byDate.set(date, set);
};
// Trace directory: the date directory name is the local date (yyyy-mm-dd) that core uses when writing to disk.
const dir = tracesDir(this.deps.root, projectId, agentId);
for (const dateDir of await listDirsSafe(dir)) {
for (const file of await listFilesSafe(path.join(dir, dateDir))) {
const match = TRACE_FILE_RE.exec(file);
if (match) mark(dateDir, match[1]!);
}
}
// DB index: the creation day also counts as active (a Session that hasn't run a Task yet produces no Trace).
for (const row of this.deps.sessions.listByAgent(projectId, agentId)) {
const created = new Date(row.createdAt);
if (Number.isNaN(created.getTime())) all.add(row.sessionId);
else mark(localDate(created), row.sessionId);
}
const activity: number[] = [];
const now = new Date();
for (let i = days - 1; i >= 0; i--) {
const d = new Date(now.getFullYear(), now.getMonth(), now.getDate() - i);
activity.push(byDate.get(localDate(d))?.size ?? 0);
}
return { sessionCount: all.size, activity };
}
/**
* Create a Session: model reference `(provider, modelId)` as a pair; defaults to
* the Project's default reference (400 prompting to configure a model first if
* there is none); when provider is omitted, core's resolveModelRef performs
* unique resolution (400 on zero matches / ambiguity). `workspace` is already
* validated by the route guard. The new Session is added to the active table
* (idle).
*/
async createSession(args: {
projectId: string;
agentId: string;
/** Upstream id of the session's model (paired with provider); defaults to the Project's default reference. */
modelId?: string;
/** The provider group for `modelId`; if omitted, resolveModelRef performs unique resolution. */
provider?: string;
workspace?: string;
approvalMode?: ApprovalMode;
/** Session source marker (schedule when triggered by a scheduled task; defaults to user-created). */
source?: "schedule";
}): Promise<SessionInfo> {
let modelId = args.modelId;
let provider = args.provider;
if (modelId === undefined) {
const def = await this.deps.projectConfig.getDefaultModelRef(args.projectId);
if (def === undefined) {
throw new HttpError(
400,
"no_default_model",
"该 Project 尚未配置默认模型,请先在「模型」页添加模型并设为默认。",
);
}
modelId = def.model_id;
provider = def.provider;
}
const agent = await createAgent({
root: this.deps.root,
projectId: args.projectId,
agentId: args.agentId,
});
let session;
try {
session = await agent.createSession({
modelId,
...(provider !== undefined ? { provider } : {}),
...(args.workspace !== undefined ? { workspaceDir: args.workspace } : {}),
});
} catch (err) {
// A missing credential is its own category (the frontend shows localized text
// by code); other core errors (zero matches / ambiguous reference, Workspace
// not existing, etc.) are collapsed to 400 — the guard already blocks most cases.
if (isMissingCredential(err)) throw modelCredentialMissing(modelId);
throw new HttpError(
400,
"session_create_failed",
err instanceof Error ? err.message : String(err),
);
}
const row: SessionRow = {
sessionId: session.sessionId,
projectId: args.projectId,
agentId: args.agentId,
provider: session.provider,
modelId: session.modelId,
workspace: session.workspaceDir,
approvalMode: args.approvalMode ?? "allow-all",
title: null,
createdAt: new Date().toISOString(),
...(args.source !== undefined ? { source: args.source } : {}),
};
this.deps.sessions.insert(row);
this.deps.manager.adopt(row, session);
return this.toInfo(row, false);
}
/** Scans the Trace directory to get the set of session_ids with records. */
private async discoverTraceSessionIds(projectId: string, agentId: string): Promise<Set<string>> {
const dir = tracesDir(this.deps.root, projectId, agentId);
const ids = new Set<string>();
for (const dateDir of await listDirsSafe(dir)) {
for (const file of await listFilesSafe(path.join(dir, dateDir))) {
const match = TRACE_FILE_RE.exec(file);
if (match) ids.add(match[1]!);
}
}
return ids;
}
/** Adopts a Session that exists only in the Trace directory: reads session_meta from the first line of the earliest index file. */
private async adoptTraceSession(
projectId: string,
agentId: string,
sessionId: string,
): Promise<SessionRow | null> {
const dir = tracesDir(this.deps.root, projectId, agentId);
let earliest: { path: string; index: number } | null = null;
for (const dateDir of await listDirsSafe(dir)) {
for (const file of await listFilesSafe(path.join(dir, dateDir))) {
const match = TRACE_FILE_RE.exec(file);
if (!match || match[1] !== sessionId) continue;
const index = Number(match[2]);
if (!earliest || index < earliest.index) {
earliest = { path: path.join(dir, dateDir, file), index };
}
}
}
if (!earliest) return null;
let messages;
try {
messages = await readTraceTolerant(earliest.path);
} catch {
return null; // Corrupt file: skip (does not block the list)
}
const meta = messages.find(isSessionMeta);
if (!meta) return null;
// An older Trace version's session_meta lacks provider (the model reference
// wasn't split into separate fields yet): no backward compat, skip adoption
// (core will give a clear error on resume; the product hasn't launched yet, so
// old data can simply be deleted and recreated).
if (typeof meta.payload.provider !== "string") return null;
const row: SessionRow = {
sessionId,
projectId,
agentId,
provider: meta.payload.provider,
modelId: meta.payload.model_id,
workspace: meta.payload.workspace,
// The approval mode for an unmanaged Session (started via the CLI) isn't in the Trace, so it's backfilled with the default value.
approvalMode: "allow-all",
title: null,
createdAt: sessionIdCreatedAt(sessionId) ?? meta.timestamp,
};
// Idempotent backfill: concurrent list calls may discover the same Session for the first time simultaneously (consistent with AgentsRepo's convention).
this.deps.sessions.insertOrIgnore(row);
return row;
}
}
/** Local date as yyyy-mm-dd (matches the Trace date directory convention: core's internal formatLocalDate, not publicly exported). */
function localDate(d: Date): string {
const pad = (n: number) => (n < 10 ? `0${n}` : `${n}`);
return `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())}`;
}
async function listDirsSafe(dir: string): Promise<string[]> {
try {
const entries = await readdir(dir, { withFileTypes: true });
return entries.filter((e) => e.isDirectory()).map((e) => e.name);
} catch {
return [];
}
}
async function listFilesSafe(dir: string): Promise<string[]> {
try {
const entries = await readdir(dir, { withFileTypes: true });
return entries.filter((e) => e.isFile()).map((e) => e.name);
} catch {
return [];
}
}
@@ -0,0 +1,190 @@
/**
* Agent State version snapshots and export/import.
*
* A snapshot = `snapshots/v<version>.tar.gz`, packaging `agent_state/` (archive
* entries are rooted at `agent_state/`), **excluding `.vault.toml`** (secrets never
* go into a snapshot); if a snapshot for the same version already exists, it isn't
* repacked. Import goes by the `version` inside the package and keeps the current
* vault; a snapshot of the current version is automatically taken before import;
* importing a package version equal to or lower than the current version requires
* explicit confirmation (otherwise 409).
* Docs: /docs/self-improvement § "Snapshots and versions".
*/
import { randomBytes } from "node:crypto";
import fs from "node:fs/promises";
import path from "node:path";
import * as tar from "tar";
import { parse as parseYaml } from "yaml";
import {
agentDir,
agentStateDir,
agentStateVersion,
agentVaultPath,
snapshotsDir,
systemConfigPath,
} from "@prismshadow/penguin-core";
import { HttpError } from "../http/errors.js";
import { badRequest } from "../http/validate.js";
/** Vault file name inside a snapshot/import package (used for archive filtering). */
const VAULT_BASENAME = ".vault.toml";
function isVaultEntry(entryPath: string): boolean {
return path.posix.basename(entryPath.replaceAll("\\", "/")) === VAULT_BASENAME;
}
export class SnapshotService {
constructor(private readonly root: string) {}
/** Current Agent State version number (missing field treated as 1); throws 404 if the Agent doesn't exist. */
async currentVersion(projectId: string, agentId: string): Promise<number> {
let raw: string;
try {
raw = await fs.readFile(systemConfigPath(this.root, projectId, agentId), "utf8");
} catch {
throw new HttpError(404, "agent_not_found", "Agent 不存在。");
}
const parsed = parseYaml(raw) as { version?: unknown } | null;
return agentStateVersion({
version: typeof parsed?.version === "number" ? parsed.version : undefined,
});
}
/** Ensures a snapshot exists for the current version (not repacked for the same version), returns the snapshot file path and version number. */
async ensureSnapshot(
projectId: string,
agentId: string,
): Promise<{ version: number; file: string }> {
const version = await this.currentVersion(projectId, agentId);
const dir = snapshotsDir(this.root, projectId, agentId);
const file = path.join(dir, `v${version}.tar.gz`);
try {
await fs.access(file);
return { version, file };
} catch {
// No snapshot for this version yet: pack it.
}
await fs.mkdir(dir, { recursive: true });
const tmp = `${file}.tmp-${randomBytes(4).toString("hex")}`;
await tar.create(
{
gzip: true,
cwd: agentDir(this.root, projectId, agentId),
file: tmp,
portable: true,
filter: (p) => !isVaultEntry(p),
},
["agent_state"],
);
await fs.rename(tmp, file);
return { version, file };
}
/** Export: automatically packs a snapshot first if none exists, returns the info needed for download. */
async exportArchive(
projectId: string,
agentId: string,
): Promise<{ version: number; file: string; fileName: string }> {
const { version, file } = await this.ensureSnapshot(projectId, agentId);
return { version, file, fileName: `${agentId}-v${version}.tar.gz` };
}
/**
* Import: validates the package structure and `version`, compares versions
* (higher than current imports directly, same version or older requires
* `confirm`), automatically snapshots the current version before import, then
* replaces `agent_state/` while keeping the current vault.
*/
async importArchive(
projectId: string,
agentId: string,
archive: Buffer,
confirm: boolean,
): Promise<{ version: number }> {
const current = await this.currentVersion(projectId, agentId);
const base = agentDir(this.root, projectId, agentId);
const staging = path.join(base, `.import-${randomBytes(6).toString("hex")}`);
await fs.mkdir(staging, { recursive: true });
try {
const archiveFile = path.join(staging, "archive.tar.gz");
await fs.writeFile(archiveFile, archive);
const extractDir = path.join(staging, "extracted");
await fs.mkdir(extractDir, { recursive: true });
try {
await tar.extract({
file: archiveFile,
cwd: extractDir,
filter: (p) => !isVaultEntry(p),
});
} catch {
throw badRequest("导入失败:不是合法的 tar.gz 快照包。");
}
// Validation: the package must contain agent_state/system_config.yaml, and version must be valid.
const configPath = path.join(extractDir, "agent_state", "system_config.yaml");
let parsed: unknown;
try {
parsed = parseYaml(await fs.readFile(configPath, "utf8"));
} catch {
throw badRequest("导入失败:包内缺少 agent_state/system_config.yaml。");
}
if (
parsed === null ||
typeof parsed !== "object" ||
typeof (parsed as { system_prompt?: unknown }).system_prompt !== "string"
) {
throw badRequest("导入失败:包内 system_config.yaml 非法。");
}
const incomingRaw = (parsed as { version?: unknown }).version;
if (
incomingRaw !== undefined &&
(!Number.isInteger(incomingRaw) || (incomingRaw as number) < 1)
) {
throw badRequest("导入失败:包内 version 非法。");
}
const incoming = agentStateVersion({
version: typeof incomingRaw === "number" ? incomingRaw : undefined,
});
if (incoming <= current && !confirm) {
throw new HttpError(
409,
"version_conflict",
`包版本 v${incoming} 不高于当前 v${current},需要确认后导入。`,
);
}
// Automatically snapshots the current version before import (reused if one already exists for that version), so a mistaken import can be rolled back.
await this.ensureSnapshot(projectId, agentId);
// Replace agent_state: first merge the current vault into the staging
// directory to be swapped in (extraction already filtered out any vault
// inside the package, so no conflict), making the replacement a pure rename
// swap — once the swap lands, there's no further write that can fail. The
// vault never goes into a snapshot, so if recovery fails after the swap, the
// `finally` cleanup of staging would delete the only vault copy with no way
// to roll back.
const stateDir = agentStateDir(this.root, projectId, agentId);
const incomingState = path.join(extractDir, "agent_state");
let vault: Buffer | null = null;
try {
vault = await fs.readFile(agentVaultPath(this.root, projectId, agentId));
} catch {
// No vault means nothing to preserve.
}
if (vault !== null) {
await fs.writeFile(path.join(incomingState, VAULT_BASENAME), vault, { mode: 0o600 });
}
const trash = path.join(staging, "replaced-agent_state");
await fs.rename(stateDir, trash);
try {
await fs.rename(incomingState, stateDir);
} catch (err) {
await fs.rename(trash, stateDir); // Rollback: restore the old directory if swapping in the new one fails.
throw err;
}
return { version: incoming };
} finally {
await fs.rm(staging, { recursive: true, force: true });
}
}
}
@@ -0,0 +1,683 @@
/**
* Trace service.
*
* History messages: all of the Session's index files concatenated in order
* (readTraceTolerant, tolerating a truncated last line), containing only the
* complete messages and events that were actually written to Trace (naturally
* excluding partial_*); in-flight increments are continued by SSE.
* Performance analysis is derived from a single Trace file: nearest-neighbor
* pairing of request_begin/end, tool call duration pairing, reconnect / compaction
* counts, and Token trend.
*/
import fs from "node:fs/promises";
import path from "node:path";
import { agentsDir, readTraceTolerant, tracesDir } from "@prismshadow/penguin-core";
import type { OmniMessage } from "@prismshadow/penguin-core";
import type {
AgentTracesResponse,
RequestSpan,
ToolCallSpan,
TraceAnalysisResponse,
TraceEventsResponse,
TraceFileInfo,
TraceModelSegment,
TraceTaskStats,
TraceToolSpan,
UsageTrendPointInTrace,
} from "../api/types.js";
import { HttpError } from "../http/errors.js";
const TRACE_FILE_RE = /^(.+)_(\d{3})\.jsonl$/;
/** Recursion depth cap for sub-session expansion (run_subagent depth is already constrained by the SDK; this is just a defensive backstop against cycles). */
const MAX_SUBAGENT_DEPTH = 4;
interface LocatedFile {
path: string;
date: string;
index: number;
}
/**
* A **direct sub-session pointer** (the `subagent` event in the parent Trace) ->
* the sub-session's Session id. The pointer only
* records the Session id; the sub-session's Agent is located within the Project by
* its Trace file.
*/
function subagentPointer(msg: OmniMessage): string | null {
if (msg.type !== "event_msg") return null;
const p = msg.payload as { type?: string; session_id?: unknown };
if (p.type !== "subagent" || typeof p.session_id !== "string" || p.session_id === "") {
return null;
}
return p.session_id;
}
async function listDirs(dir: string): Promise<string[]> {
try {
const entries = await fs.readdir(dir, { withFileTypes: true });
return entries.filter((e) => e.isDirectory()).map((e) => e.name);
} catch {
return [];
}
}
async function listFiles(dir: string): Promise<string[]> {
try {
const entries = await fs.readdir(dir, { withFileTypes: true });
return entries.filter((e) => e.isFile()).map((e) => e.name);
} catch {
return [];
}
}
export class TraceService {
constructor(private readonly root: string) {}
/** All of this Session's Trace files (sorted by index ascending). */
private async locateAll(
projectId: string,
agentId: string,
sessionId: string,
): Promise<LocatedFile[]> {
const dir = tracesDir(this.root, projectId, agentId);
const out: LocatedFile[] = [];
for (const dateDir of await listDirs(dir)) {
for (const file of await listFiles(path.join(dir, dateDir))) {
const match = TRACE_FILE_RE.exec(file);
if (!match || match[1] !== sessionId) continue;
out.push({ path: path.join(dir, dateDir, file), date: dateDir, index: Number(match[2]) });
}
}
return out.sort((a, b) => a.index - b.index);
}
/** Deletes all of this Session's Trace files (called when the Session is deleted). */
async deleteSessionTraces(projectId: string, agentId: string, sessionId: string): Promise<void> {
const files = await this.locateAll(projectId, agentId, sessionId);
for (const file of files) {
await fs.rm(file.path, { force: true });
}
}
/**
* History messages: all index files concatenated in order, with sub-sessions
* **expanded in place**.
*
* The parent Trace only records a `subagent` pointer event at the spawn point
* (recording just the child Session id; the content lives in the child
* Session's own Trace). Here the pointer is used to locate the child Trace
* within the Project, read it recursively, and splice the child messages —
* tagged with an origin chain — back in at the pointer's position, so that when
* the session is reopened, the frontend can re-attach the sub-session to the
* run_subagent tool card via origin (the child Trace's first `session_meta`,
* once given an origin, takes the same shape as what's forwarded over the live
* stream). When expansion succeeds, the pointer event itself is no longer
* emitted; when the child Trace is missing (deleted), the pointer event is kept
* so API consumers can still know it existed.
*/
async readMessages(
projectId: string,
agentId: string,
sessionId: string,
): Promise<OmniMessage[]> {
return this.readMessagesExpanded(projectId, agentId, sessionId, {
index: null,
ancestry: new Set([sessionId]),
depth: 0,
});
}
/**
* A Project-wide session location index (sessionId -> agentId): built by
* scanning every Agent's traces directory. Built lazily the first time a
* subagent pointer is encountered, then reused across the whole readMessages
* call — rescanning per pointer would blow up into tens of thousands of readdir
* calls under multiple sub-sessions plus recursive expansion.
*/
private async buildSessionIndex(projectId: string): Promise<Map<string, string>> {
const index = new Map<string, string>();
for (const agentId of await listDirs(agentsDir(this.root, projectId))) {
const dir = tracesDir(this.root, projectId, agentId);
for (const dateDir of await listDirs(dir)) {
for (const file of await listFiles(path.join(dir, dateDir))) {
const match = TRACE_FILE_RE.exec(file);
if (match && !index.has(match[1]!)) index.set(match[1]!, agentId);
}
}
}
return index;
}
private async readMessagesExpanded(
projectId: string,
agentId: string,
sessionId: string,
ctx: { index: Map<string, string> | null; ancestry: Set<string>; depth: number },
): Promise<OmniMessage[]> {
const files = await this.locateAll(projectId, agentId, sessionId);
const out: OmniMessage[] = [];
for (const file of files) {
for (const msg of await readTraceTolerant(file.path)) {
// The depth cap guards against runaway recursion; ancestry guards against a
// cyclic pointer (a tampered Trace pointing to itself/an ancestor is not expanded).
const childSid = ctx.depth < MAX_SUBAGENT_DEPTH ? subagentPointer(msg) : null;
if (!childSid || ctx.ancestry.has(childSid)) {
out.push(msg);
continue;
}
ctx.index ??= await this.buildSessionIndex(projectId);
const childAgent = ctx.index.get(childSid);
let nested: OmniMessage[] = [];
if (childAgent) {
ctx.ancestry.add(childSid);
nested = await this.readMessagesExpanded(projectId, childAgent, childSid, {
...ctx,
depth: ctx.depth + 1,
});
ctx.ancestry.delete(childSid);
}
// Child Trace missing (deleted): keep the pointer event, since the sub-session's content can't be recovered.
if (nested.length === 0) {
out.push(msg);
continue;
}
for (const m of nested) out.push({ ...m, origin: [childSid, ...(m.origin ?? [])] });
}
}
return out;
}
/** List of Trace files (index / date / size / mtime). */
async listTraceFiles(
projectId: string,
agentId: string,
sessionId: string,
): Promise<TraceFileInfo[]> {
const files = await this.locateAll(projectId, agentId, sessionId);
const out: TraceFileInfo[] = [];
for (const file of files) {
const stat = await fs.stat(file.path);
out.push({
index: file.index,
date: file.date,
sizeBytes: stat.size,
mtime: stat.mtime.toISOString(),
});
}
return out;
}
/** Reads events from the Trace file at the given index, paginated by line (for loading large files in pages). */
async readEvents(
projectId: string,
agentId: string,
sessionId: string,
index: number,
offset: number,
limit: number,
): Promise<TraceEventsResponse> {
const messages = await this.readFileByIndex(projectId, agentId, sessionId, index);
return {
events: messages.slice(offset, offset + limit),
offset,
limit,
total: messages.length,
};
}
/** Performance analysis: derived from a single Trace file. */
async analyze(
projectId: string,
agentId: string,
sessionId: string,
index: number,
): Promise<TraceAnalysisResponse> {
const messages = await this.readFileByIndex(projectId, agentId, sessionId, index);
const requests: RequestSpan[] = [];
let openRequest: RequestSpan | null = null;
const toolCalls: ToolCallSpan[] = [];
const openToolCalls = new Map<string, ToolCallSpan>();
let reconnectCount = 0;
let compactionCount = 0;
const usageTrend: UsageTrendPointInTrace[] = [];
// Timeline (serial-duration estimation): Trace records completion times; model
// messages are produced
// serially (autoregressive decoding), so each segment's start = the previous
// event's time (the request's first segment = request_begin); a tool's
// approval/execution runs in parallel with model decoding, on its own lane;
// prevSerialTs is cleared after request_end, and the next request_begin
// restarts the count (which presumes all of the previous round's
// tool_call_output have already come back).
const modelSegments: TraceModelSegment[] = [];
const toolSpans: TraceToolSpan[] = [];
const openSpansById = new Map<string, TraceToolSpan>();
let prevSerialTs: string | null = null;
// Task grouping: one user turn contains multiple Request rounds (the Agent
// loop sends another round each time it calls a tool); the turn ends once the
// model produces only text with no further tool call. Consecutive Requests are
// merged into one Task on this basis, and each Task gets its own independent
// timeline — different Tasks can be far apart in time (the user is thinking or
// has stepped away), and sharing one timeline would leave large gaps.
// Compaction forms its own turn: both compaction_begin/compaction_end break a
// continuation, so the compaction request becomes its own Task, and the
// request that resumes after compaction starts yet another Task.
let taskIndex = -1;
let continuation = false; // The previous round's Request called a tool -> the next request_begin continues the same Task
let sawToolCallThisRequest = false;
// Compaction interval (compaction_begin..compaction_end): the compaction
// request's request_begin/request_end and token_usage all fall inside it (see
// core context-engine's summarize flow), which is used to exclude the
// compaction request entirely from TPS — matching the same convention as
// compactionActive in the Chat page's task-stats.
let compactionActive = false;
// Token / duration totals per Task (computed server-side over the whole file:
// frontend events are fetched in pages, so summing them there would be
// mismatched).
const taskStats = new Map<number, TraceTaskStats>();
const ensureTask = (ti: number): TraceTaskStats => {
let t = taskStats.get(ti);
if (t === undefined) {
t = {
taskIndex: ti,
messageFrom: -1,
messageTo: -1,
startTs: "",
endTs: "",
tokens: { cacheRead: 0, cacheWrite: 0, output: 0 },
llmMs: 0,
};
taskStats.set(ti, t);
}
return t;
};
/**
* Which turn each message belongs to: **decided definitively in one
* sequential pass**, not left for the frontend to guess by timestamp.
*
* Timestamp boundaries can't be pulled apart — the same millisecond can
* contain "the previous turn's last reply, compaction_begin, the compaction
* prompt, and the next turn's request_begin" all at once, so assigning by
* time would inevitably misfile this turn's reply into the next turn.
*
* Rule (a turn = one user turn; `request_end` is the end of some Request
* within a turn):
* - The **starting marker** of a new turn: the main session's user Prompt
* (outside compaction), or compaction_begin (compaction forms its own
* turn). Messages after the marker and before that turn's first
* `request_begin` (subsequent images from a multi-image send, the
* compaction prompt) are always held pending, waiting for
* `request_begin` to settle the new taskIndex before the whole span is
* assigned at once — they belong to the **new** turn, not the tail of
* the previous one.
* - Other messages belong to the current taskIndex: tool output and
* approval decisions arriving after request_end still belong to this
* turn (they're the results of tools this turn's Request initiated).
*/
const msgTask: number[] = new Array<number>(messages.length).fill(-1);
/** The pending new turn's starting point (message index); settled once request_begin determines the taskIndex. */
let pendingFrom: number | null = null;
for (let mi = 0; mi < messages.length; mi++) {
const msg = messages[mi]!;
const p = msg.payload as Record<string, unknown> & { type?: string };
// The timeline only looks at the main session (a Trace itself never contains origin messages; this is a defensive skip).
const hasOrigin = msg.origin !== undefined && msg.origin.length > 0;
// Starting marker of a new turn: the main session's user Prompt (outside
// compaction) -> a new user turn; compaction_begin -> a compaction turn
// (compaction forms its own turn). A single send can be "text + multiple
// images" = multiple messages; only the **first** one counts (once
// pendingFrom is set, it's not changed again), otherwise the turn's start
// would shift to the last image.
const startsUserTurn =
!hasOrigin &&
!compactionActive &&
msg.type === "model_msg" &&
((p.type === "text" && p.role === "user") || p.type === "image_url");
const startsCompactionTurn =
!hasOrigin && msg.type === "event_msg" && p.type === "compaction_begin";
if (startsUserTurn || startsCompactionTurn) {
if (pendingFrom === null) pendingFrom = mi;
// A user Prompt **always starts a new turn**: judging continuation solely
// by "did the previous turn call a tool" isn't enough — if the previous
// turn ended in timeout/malformed (given up after exhausting retries),
// retryable would leave continuation at true, and this new message would
// get merged into that failed turn, smearing the two turns' messages /
// Tokens / TPS / duration together.
if (startsUserTurn) continuation = false;
}
// A main-session message that isn't pending belongs to the current turn immediately (taskIndex < 0 = before the first request_begin, e.g. session_meta).
if (!hasOrigin && pendingFrom === null) msgTask[mi] = taskIndex;
if (msg.type === "event_msg") {
if (p.type === "request_begin") {
if (!hasOrigin) {
prevSerialTs = msg.timestamp;
if (!continuation) taskIndex++; // Not a continuation -> a new Task
sawToolCallThisRequest = false;
}
// Settle taskIndex before opening the span: the span belongs directly to
// the current Task. Nearest-neighbor pairing: if the previous begin was
// never closed (process exited mid-run), the span is left open.
openRequest = { beginTs: msg.timestamp, taskIndex };
if (compactionActive) openRequest.compaction = true;
requests.push(openRequest);
if (!hasOrigin) {
const t = ensureTask(taskIndex);
if (compactionActive) t.compaction = true; // This turn is a compaction turn
// This turn's duration starts at the first request_begin. It doesn't
// use the timestamp of the user Prompt / compaction summary or other
// user text: `<context_summary>` is created during compaction but only
// written to disk on the next run, so resuming the next day would
// stretch the first turn out by a whole day for no reason; the Prompt
// to request-dispatch gap is only ever milliseconds anyway.
if (t.startTs === "") t.startTs = msg.timestamp;
// The new turn's taskIndex is only settled here: the pending span
// (user Prompt / multiple images / compaction prompt) is assigned in
// full to **this** turn — they're the start of the new turn, not the
// tail of the previous one.
if (pendingFrom !== null) {
for (let k = pendingFrom; k < mi; k++) {
if (messages[k]!.origin === undefined) msgTask[k] = taskIndex;
}
pendingFrom = null;
}
msgTask[mi] = taskIndex;
}
} else if (p.type === "approval_decision") {
if (!hasOrigin && typeof p.tool_call_id === "string") {
const span = openSpansById.get(p.tool_call_id);
if (span && span.approvalTs === undefined) {
span.approvalTs = msg.timestamp;
if (typeof p.decision === "string") span.decision = p.decision;
// Approval wait time is subtracted out of the LLM generation
// duration: core does `await approve(tc)` inside the streaming loop,
// so the entire manual wait sits between request_begin and
// request_end (see RequestSpan.approvalWaitMs). Without subtracting
// it, "5s generation + 55s approval wait" would show 100 tok/s as 8 tok/s.
if (openRequest) {
const wait = Date.parse(msg.timestamp) - Date.parse(span.callTs);
if (Number.isFinite(wait) && wait > 0) {
openRequest.approvalWaitMs = (openRequest.approvalWaitMs ?? 0) + wait;
}
}
}
}
} else if (p.type === "request_end") {
const status = typeof p.status === "string" ? p.status : undefined;
// timeout/malformed is automatically reconnected by core within the same
// run (context-engine's retry loop); the resent Request still belongs to
// **the same user turn**: it must continue the turn, otherwise a single
// timeout would split that turn's Tokens/duration/TPS across two Tasks.
const retryable = status === "timeout" || status === "malformed";
if (!hasOrigin) {
prevSerialTs = null;
continuation = sawToolCallThisRequest || retryable;
}
if (retryable) reconnectCount++;
if (openRequest) {
openRequest.endTs = msg.timestamp;
const dur = Date.parse(msg.timestamp) - Date.parse(openRequest.beginTs);
if (Number.isFinite(dur)) {
openRequest.durationMs = dur;
openRequest.activeMs = Math.max(0, dur - (openRequest.approvalWaitMs ?? 0));
}
if (status !== undefined) openRequest.status = status;
// TPS denominator: accumulated per the turn a Request belongs to. A
// compaction request counts too — it belongs to **its own compaction
// turn** (compaction forms its own turn), so it neither pollutes a
// user turn's TPS, nor does the compaction turn fail to report its own
// generation speed accurately. A failed retry's duration is counted as
// well — it belongs to the same turn as the retry that eventually
// succeeded, and "how long this turn took to produce these tokens"
// should include the retries by definition.
if (!hasOrigin && openRequest.activeMs !== undefined) {
ensureTask(openRequest.taskIndex).llmMs += openRequest.activeMs;
}
openRequest = null;
}
} else if (p.type === "compaction_begin") {
compactionCount++;
// Compaction forms its own turn: otherwise, if the previous turn called
// a tool, continuation would still be true and the compaction request
// would get merged into the previous Task.
if (!hasOrigin) {
continuation = false;
compactionActive = true;
}
} else if (p.type === "compaction_end") {
// Both ends of compaction break a continuation. This closing one can't
// be skipped: if the compaction request itself exhausts its retries and
// ends in timeout, the retryable check above would mark it as "continued",
// and without clearing it here, the next user turn after compaction
// would get merged into this compaction Task.
if (!hasOrigin) {
continuation = false;
compactionActive = false;
}
} else if (p.type === "token_usage") {
const request = p.request as
| { total?: number; cache_read?: number; cache_write?: number; output?: number }
| undefined;
const session = p.session as { total?: number } | undefined;
usageTrend.push({
ts: msg.timestamp,
requestTotal: request?.total ?? 0,
sessionTotal: session?.total ?? 0,
});
if (!hasOrigin) {
const t = ensureTask(taskIndex);
// Cumulative usage for this turn (a running total): those tokens were
// actually paid for, so the cost can't be dropped. `tokens.output` also
// doubles as the numerator for output TPS — compaction's output
// belongs to **its own compaction turn** (compaction forms its own
// turn), so a user turn's TPS isn't polluted by it, while the
// compaction turn can still accurately report "how fast the summary
// was generated".
t.tokens.cacheRead += request?.cache_read ?? 0;
t.tokens.cacheWrite += request?.cache_write ?? 0;
t.tokens.output += request?.output ?? 0;
if (!compactionActive) {
// The context snapshot only takes non-compaction Requests: tokens
// consumed by compaction aren't the post-compaction context
// footprint. A later write overwrites an earlier one -> this
// naturally leaves behind the snapshot of the Task's **last**
// non-compaction Request = the context footprint at the end of this
// turn. Accumulating would be wrong: each Request's input carries
// the entire history afresh (see TraceTaskStats).
t.context = {
cacheRead: request?.cache_read ?? 0,
cacheWrite: request?.cache_write ?? 0,
output: request?.output ?? 0,
};
}
}
}
continue;
}
if (msg.type !== "model_msg") continue;
// Model serial segments: assistant-side thinking/text/tool_call (a user input sent instantaneously occupies no segment).
if (
!hasOrigin &&
prevSerialTs !== null &&
(p.type === "thinking" ||
p.type === "tool_call" ||
(p.type === "text" && p.role === "assistant"))
) {
const segment: TraceModelSegment = {
kind: p.type === "thinking" ? "thinking" : p.type === "tool_call" ? "tool_call" : "text",
startTs: prevSerialTs,
endTs: msg.timestamp,
taskIndex,
};
if (p.type === "tool_call" && typeof p.tool_call_id === "string") {
segment.toolCallId = p.tool_call_id;
if (typeof p.name === "string") segment.name = p.name;
}
modelSegments.push(segment);
prevSerialTs = msg.timestamp;
}
if (p.type === "tool_call" && typeof p.tool_call_id === "string") {
if (!hasOrigin) sawToolCallThisRequest = true; // This turn called a tool -> the next turn continues the same Task
const callStop = typeof p.stop_reason === "string" ? p.stop_reason : undefined;
const span: ToolCallSpan = {
toolCallId: p.tool_call_id,
name: typeof p.name === "string" ? p.name : "",
startTs: msg.timestamp,
};
if (callStop !== undefined) span.stopReason = callStop;
openToolCalls.set(p.tool_call_id, span);
toolCalls.push(span);
// The timeline lane only accepts calls that "will actually be executed":
// an interrupt-compensation tool_call (stop_reason other than completed)
// never gets an approval/output, and putting it on a lane would render as
// a phantom "executing" state spanning the whole timeline — so it's
// skipped outright.
if (!hasOrigin && (callStop === undefined || callStop === "completed")) {
const timeline: TraceToolSpan = {
toolCallId: p.tool_call_id,
name: typeof p.name === "string" ? p.name : "",
callTs: msg.timestamp,
taskIndex,
};
openSpansById.set(p.tool_call_id, timeline);
toolSpans.push(timeline);
}
} else if (p.type === "tool_call_output" && typeof p.tool_call_id === "string") {
const span = openToolCalls.get(p.tool_call_id);
if (span && span.endTs === undefined) {
span.endTs = msg.timestamp;
const dur = Date.parse(msg.timestamp) - Date.parse(span.startTs);
if (Number.isFinite(dur)) span.durationMs = dur;
if (typeof p.stop_reason === "string") span.stopReason = p.stop_reason;
}
const timeline = openSpansById.get(p.tool_call_id);
if (timeline && timeline.outputTs === undefined) {
timeline.outputTs = msg.timestamp;
if (typeof p.stop_reason === "string") timeline.stopReason = p.stop_reason;
}
}
}
// A pending span that never got a request_begin (interrupted right after the
// user sent it / the process exited): it's a turn that never got to run,
// and forms its own turn — reattaching it to the previous turn would smear
// two separate user sends together.
if (pendingFrom !== null) {
taskIndex++;
for (let k = pendingFrom; k < messages.length; k++) {
if (messages[k]!.origin === undefined) msgTask[k] = taskIndex;
}
ensureTask(taskIndex);
}
// Each turn's message index range and end-of-turn time are always derived from
// the per-message assignment done above (same source, so they never disagree
// with each other). Messages before the first request_begin (session_meta)
// have taskIndex -1 and are assigned to the first turn, otherwise they'd have
// nowhere to sit on the page. The turn duration's **starting point** isn't
// decided here — it was already settled at request_begin (duration only looks
// at LLM requests; timestamps of the user Prompt / compaction summary or other
// user text don't participate, see TraceTaskStats.startTs).
const firstTask = [...taskStats.keys()].sort((a, b) => a - b)[0];
for (let k = 0; k < messages.length; k++) {
let ti = msgTask[k]!;
if (ti < 0) {
if (firstTask === undefined) continue;
ti = firstTask;
msgTask[k] = ti;
}
const t = ensureTask(ti);
if (t.messageFrom < 0 || k < t.messageFrom) t.messageFrom = k;
if (k > t.messageTo) t.messageTo = k;
// session_meta is only **listed** in the first turn, and doesn't count
// toward the end-of-turn time: it's metadata written when the session was
// created, and its timestamp has nothing to do with this turn (it also gets
// rewritten verbatim at the start of a new file after compaction splits the file).
if (messages[k]!.type === "session_meta") continue;
const ts = messages[k]!.timestamp;
if (t.endTs === "" || ts > t.endTs) t.endTs = ts;
}
const tasks = [...taskStats.values()].sort((a, b) => a.taskIndex - b.taskIndex);
// Total elapsed time = **the sum of each turn's duration**, matching exactly
// the scope shown per-turn below (**including compaction turns** — their wall
// clock time genuinely elapsed, each turn's card has its own duration, and the
// overall total is their sum, so the numbers must add up). It is not "last
// message timestamp minus first message timestamp": that would be the whole
// file's wall-clock span, counting in the gaps **between** turns (the user
// thinking, stepping out for coffee, coming back the next day) — none of which
// is time the Agent spent working. A degenerate turn with no Request has an
// empty startTs and counts as 0.
// Note this uses a different convention from the Session's cumulative elapsed
// time on the Chat page: that one only accumulates user turns (compaction
// after a turn ends doesn't count toward the turn).
const elapsedMs = tasks.reduce((sum, t) => {
const span = Date.parse(t.endTs) - Date.parse(t.startTs);
return sum + (Number.isFinite(span) ? Math.max(0, span) : 0);
}, 0);
return {
elapsedMs,
requests,
tasks,
toolCalls,
modelSegments,
toolSpans,
reconnectCount,
compactionCount,
usageTrend,
};
}
/** Level-by-level browsing (newest first): Agent -> date -> Session -> Trace files. */
async agentTraces(projectId: string, agentId: string): Promise<AgentTracesResponse> {
const dir = tracesDir(this.root, projectId, agentId);
const dates = (await listDirs(dir)).sort().reverse();
const out: AgentTracesResponse = { dates: [] };
for (const date of dates) {
const bySession = new Map<string, { index: number; sizeBytes: number }[]>();
for (const file of await listFiles(path.join(dir, date))) {
const match = TRACE_FILE_RE.exec(file);
if (!match) continue;
const sessionId = match[1]!;
const stat = await fs.stat(path.join(dir, date, file));
const files = bySession.get(sessionId) ?? [];
files.push({ index: Number(match[2]), sizeBytes: stat.size });
bySession.set(sessionId, files);
}
if (bySession.size === 0) continue;
out.dates.push({
date,
// session_id embeds a timestamp, so reverse lexicographic order is reverse chronological order.
sessions: [...bySession.entries()]
.sort((a, b) => b[0].localeCompare(a[0]))
.map(([sessionId, files]) => ({
sessionId,
files: files.sort((a, b) => a.index - b.index),
})),
});
}
return out;
}
private async readFileByIndex(
projectId: string,
agentId: string,
sessionId: string,
index: number,
): Promise<OmniMessage[]> {
const files = await this.locateAll(projectId, agentId, sessionId);
const file = files.find((f) => f.index === index);
if (!file) {
throw new HttpError(
404,
"trace_not_found",
`该 Session 没有 index 为 ${index} 的 Trace 文件。`,
);
}
return readTraceTolerant(file.path);
}
}
@@ -0,0 +1,269 @@
/**
* Usage statistics query.
*
* Cost is **computed in real time**: usage_records only stores Tokens (pricing may
* be added later), so at query time each Model's cost is converted using the
* current Project's configured pricing — the repo returns raw Token totals broken
* down by `(provider, model_id)` paired reference, and this service looks up each
* reference's price once and folds it into cost / hasUncosted (if a Model has no
* pricing, its consumption is excluded from cost and hasUncosted is flagged).
* Summary cards (today / last 7 days / cumulative), grouped aggregation (date /
* agent / model / session, with the session dimension supporting agentId drill-down
* filtering), and a 30-day trend.
* Server-side error statistics (error_records) ride along on the same response:
* the statistics center fetches everything in one request, and filters are
* naturally shared; unattributed errors (login failures, process crashes, and other
* errors with no Project context) are visible only to admins, see the ErrorsRepo
* file header.
*/
import type {
UsageBucket,
UsageErrors,
UsageGroupBy,
UsageGroupRow,
UsageResponse,
} from "../api/types.js";
import type { ErrorFilter, ErrorsRepo } from "../db/repos/errors.js";
import type {
UsageRepo,
UsageModelSums,
UsageGroupModelSums,
UsageFilter,
} from "../db/repos/usage.js";
import { formatLocalDate, localDateMinusDays } from "../internal/dates.js";
/** Number of most-recent entries kept in the error detail table. */
const ERROR_RECENT_N = 20;
/** The three pricing buckets (usd_per_mtok convention), returned by the pricing lookup callback. */
export interface PricingRates {
cacheRead: number;
cacheWrite: number;
output: number;
}
export type PricingLookup = (
projectId: string,
provider: string,
modelId: string,
) => Promise<PricingRates | undefined>;
export interface UsageQuery {
from?: string;
to?: string;
groupBy: UsageGroupBy;
/** Top-level filter: view by Agent (also used for groupBy=session drill-down). */
agentId?: string;
/** Top-level filter: view by Model (paired with modelId; the dropdown always sends them as a pair). */
provider?: string;
modelId?: string;
/** Whether to include unattributed errors: admin only (the route passes user.isAdmin), defaults to false. */
includeGlobalErrors?: boolean;
}
/** Cost formula: sum of the three buckets, in USD per million Tokens. */
function costOf(sums: UsageModelSums, rates: PricingRates): number {
return (
(sums.cacheRead * rates.cacheRead +
sums.cacheWrite * rates.cacheWrite +
sums.output * rates.output) /
1e6
);
}
/** In-process Map key for a paired reference (\0-separated, the same style as session-manager's agentKey; never persisted). */
function refKey(provider: string, modelId: string): string {
return `${provider}\0${modelId}`;
}
export class UsageService {
constructor(
private readonly usage: UsageRepo,
private readonly errors: ErrorsRepo,
private readonly lookupPricing: PricingLookup,
private readonly now: () => Date = () => new Date(),
) {}
async query(projectId: string, q: UsageQuery): Promise<UsageResponse> {
const today = formatLocalDate(this.now());
// Top-level filter: agent + model (the cost center switches views by agent/model; the model filter is always sent as a pair).
const base: UsageFilter = {};
if (q.agentId !== undefined) base.agentId = q.agentId;
if (q.provider !== undefined) base.provider = q.provider;
if (q.modelId !== undefined) base.modelId = q.modelId;
const win = (from?: string, to?: string): UsageFilter => ({
...base,
...(from !== undefined ? { from } : {}),
...(to !== undefined ? { to } : {}),
});
const todayRows = this.usage.bucketByModel(projectId, win(today, today));
const last7dRows = this.usage.bucketByModel(projectId, win(localDateMinusDays(this.now(), 6)));
const totalRows = this.usage.bucketByModel(projectId, win(q.from, q.to));
const groupRows = this.usage.groupsByModel(projectId, q.groupBy, win(q.from, q.to));
// Fixed 30-day window; affected by the agent/model filter.
const trendFrom = localDateMinusDays(this.now(), 29);
const trendRows = this.usage.groupsByModel(projectId, "date", win(trendFrom));
// Agent call-count chart: not affected by the agent filter (shows all agents), but still affected by the date + model filter.
const agentRows = this.usage.groupsByModel(projectId, "agent", {
...(q.provider !== undefined ? { provider: q.provider } : {}),
...(q.modelId !== undefined ? { modelId: q.modelId } : {}),
...(q.from !== undefined ? { from: q.from } : {}),
...(q.to !== undefined ? { to: q.to } : {}),
});
// Model success-rate chart: not affected by the model filter (shows all models), but still affected by the date + agent filter.
const statusRows = this.usage.statusByModel(projectId, {
...(q.agentId !== undefined ? { agentId: q.agentId } : {}),
...(q.from !== undefined ? { from: q.from } : {}),
...(q.to !== undefined ? { to: q.to } : {}),
});
// Error statistics: likewise not affected by the model filter (HTTP / process errors have no Model dimension), but still affected by the date + agent filter.
const errorFilter: ErrorFilter = {
...(q.agentId !== undefined ? { agentId: q.agentId } : {}),
...(q.from !== undefined ? { from: q.from } : {}),
...(q.to !== undefined ? { to: q.to } : {}),
// Unattributed errors are visible only to admins (regular members only see errors within their own Project, see the ErrorsRepo file header).
...(q.includeGlobalErrors === true ? { includeGlobal: true } : {}),
};
// Each paired reference that occurs is looked up for its current price only once.
const rates = new Map<string, PricingRates | undefined>();
const allRefs = new Map<string, { provider: string; modelId: string }>();
for (const r of [...todayRows, ...last7dRows, ...totalRows, ...groupRows, ...trendRows]) {
allRefs.set(refKey(r.provider, r.modelId), { provider: r.provider, modelId: r.modelId });
}
for (const [key, ref] of allRefs) {
rates.set(key, await this.lookupPricing(projectId, ref.provider, ref.modelId));
}
const byAgentMap = new Map<string, { requests: number; total: number }>();
for (const r of agentRows) {
const acc = byAgentMap.get(r.key) ?? { requests: 0, total: 0 };
acc.requests += r.requests;
acc.total += r.total;
byAgentMap.set(r.key, acc);
}
return {
summary: {
today: this.foldBucket(todayRows, rates),
last7d: this.foldBucket(last7dRows, rates),
total: this.foldBucket(totalRows, rates),
},
groupBy: q.groupBy,
groups: this.foldGroups(groupRows, rates, q.groupBy),
trend: this.foldTrend(trendRows, rates),
byAgent: [...byAgentMap.entries()]
.map(([agentId, v]) => ({ agentId, requests: v.requests, total: v.total }))
.sort((a, b) => b.requests - a.requests),
success: statusRows.sort((a, b) => b.total - a.total),
errors: this.foldErrors(projectId, errorFilter),
agentIds: this.usage.distinctAgentIds(projectId),
models: this.usage.distinctModels(projectId),
};
}
/** Error statistics: summary info (total / unexpected / most common error code) + the last N entries, all filtered by the selected range. */
private foldErrors(projectId: string, f: ErrorFilter): UsageErrors {
const { total, unexpected } = this.errors.summary(projectId, f);
return {
total,
unexpected,
topCode: this.errors.topCode(projectId, f),
recent: this.errors.recent(projectId, f, ERROR_RECENT_N),
};
}
private foldBucket(
rows: UsageModelSums[],
rates: Map<string, PricingRates | undefined>,
): UsageBucket {
let total = 0;
let requests = 0;
let cost: number | null = null;
let hasUncosted = false;
for (const r of rows) {
total += r.total;
requests += r.requests;
const rate = rates.get(refKey(r.provider, r.modelId));
if (rate) cost = (cost ?? 0) + costOf(r, rate);
else hasUncosted = true;
}
return { total, requests, cost, hasUncosted };
}
private foldGroups(
rows: UsageGroupModelSums[],
rates: Map<string, PricingRates | undefined>,
groupBy: UsageGroupBy,
): UsageGroupRow[] {
// The model dimension folds by paired reference (a shared model_id name across providers is split into separate rows); other dimensions fold by their group key.
const keyOf = (r: UsageGroupModelSums): string =>
groupBy === "model" ? refKey(r.provider, r.key) : r.key;
const byKey = new Map<string, UsageGroupRow>();
for (const r of rows) {
const acc = byKey.get(keyOf(r)) ?? {
key: r.key,
...(groupBy === "model" ? { provider: r.provider } : {}),
cacheRead: 0,
cacheWrite: 0,
output: 0,
total: 0,
requests: 0,
cost: null as number | null,
hasUncosted: false,
};
acc.cacheRead += r.cacheRead;
acc.cacheWrite += r.cacheWrite;
acc.output += r.output;
acc.total += r.total;
acc.requests += r.requests;
const rate = rates.get(refKey(r.provider, r.modelId));
if (rate) acc.cost = (acc.cost ?? 0) + costOf(r, rate);
else acc.hasUncosted = true;
byKey.set(keyOf(r), acc);
}
const out = [...byKey.values()];
// The date dimension sorts by key descending (most recent first); other dimensions sort by total Token count descending.
if (groupBy === "date") out.sort((a, b) => b.key.localeCompare(a.key));
else out.sort((a, b) => b.total - a.total);
return out;
}
private foldTrend(
rows: UsageGroupModelSums[],
rates: Map<string, PricingRates | undefined>,
): UsageResponse["trend"] {
const byDate = new Map<
string,
{ total: number; cacheRead: number; cacheWrite: number; output: number; cost: number | null }
>();
for (const r of rows) {
const acc = byDate.get(r.key) ?? {
total: 0,
cacheRead: 0,
cacheWrite: 0,
output: 0,
cost: null,
};
acc.total += r.total;
acc.cacheRead += r.cacheRead;
acc.cacheWrite += r.cacheWrite;
acc.output += r.output;
const rate = rates.get(refKey(r.provider, r.modelId));
if (rate) acc.cost = (acc.cost ?? 0) + costOf(r, rate);
byDate.set(r.key, acc);
}
return [...byDate.entries()]
.sort((a, b) => a[0].localeCompare(b[0]))
.map(([date, v]) => ({
date,
total: v.total,
cacheRead: v.cacheRead,
cacheWrite: v.cacheWrite,
output: v.output,
cost: v.cost,
}));
}
}
@@ -0,0 +1,283 @@
/**
* Workspace file browsing: list directory / read
* file (preview & download) / write file (upload). Security: a relative path, once
* resolved, must stay inside the Workspace — a logical prefix check plus a realpath
* check against the nearest existing ancestor (guards against `..` and symlink escapes).
*/
import fs from "node:fs/promises";
import { constants as fsc } from "node:fs";
import path from "node:path";
import type { WorkspaceFilesResponse } from "../api/types.js";
import { HttpError } from "../http/errors.js";
import { badRequest } from "../http/validate.js";
/** Per-file read cap (a safety limit since preview/download reads the whole file into memory). */
const MAX_READ_BYTES = 50 * 1024 * 1024;
/** Upload cap (stays within the 20MB request body limit even after base64 encoding). */
export const MAX_UPLOAD_BYTES = 14 * 1024 * 1024;
const CONTENT_TYPES: Record<string, string> = {
".html": "text/html; charset=utf-8",
".htm": "text/html; charset=utf-8",
".txt": "text/plain; charset=utf-8",
".md": "text/markdown; charset=utf-8",
".json": "application/json",
".js": "text/javascript; charset=utf-8",
".ts": "text/plain; charset=utf-8",
".tsx": "text/plain; charset=utf-8",
".py": "text/plain; charset=utf-8",
".sh": "text/plain; charset=utf-8",
".yaml": "text/plain; charset=utf-8",
".yml": "text/plain; charset=utf-8",
".toml": "text/plain; charset=utf-8",
".css": "text/css; charset=utf-8",
".csv": "text/plain; charset=utf-8",
".log": "text/plain; charset=utf-8",
".svg": "image/svg+xml",
".png": "image/png",
".jpg": "image/jpeg",
".jpeg": "image/jpeg",
".gif": "image/gif",
".webp": "image/webp",
".pdf": "application/pdf",
};
export interface WorkspaceFileContent {
data: Buffer;
fileName: string;
contentType: string;
/** Types whose same-origin inline rendering would execute scripts (html/svg): inline preview must fall back to plain text. */
scriptable: boolean;
}
export class WorkspaceFilesService {
/** Canonical path (realpath) of the Workspace root; 404 if it doesn't exist. */
private async realBase(workspace: string): Promise<string> {
try {
return await fs.realpath(path.resolve(workspace));
} catch {
throw new HttpError(404, "workspace_missing", "该 Session 的 Workspace 已不存在。");
}
}
/**
* Lexical containment check: whether target is inside base (including equal to
* base). Uses path.relative rather than prefix concatenation, so it works when
* base is the filesystem root ("/" concatenated with sep would produce a "//"
* prefix that no subpath could ever match); only a full ".." segment is
* compared, so a legitimate name like "..foo" isn't mistakenly rejected.
*/
private isInside(target: string, base: string): boolean {
const rel = path.relative(base, target);
return rel !== ".." && !rel.startsWith(`..${path.sep}`) && !path.isAbsolute(rel);
}
/** Lexical prefix check (a relative path, once resolved, must still be inside the Workspace); returns the absolute target path. */
private lexicalTarget(base: string, rel: string): string {
if (rel.includes("\0")) throw badRequest("path 非法。");
const target = path.resolve(base, rel === "" ? "." : rel);
if (!this.isInside(target, base)) {
throw badRequest("path 必须位于 Workspace 内。");
}
return target;
}
private assertInside(real: string, realBase: string): void {
if (!this.isInside(real, realBase)) {
throw badRequest("path 必须位于 Workspace 内。");
}
}
/**
* Read-path resolution: realpath the entire path (following all symlinks to get
* a link-free canonical path), then check containment and **perform IO on the
* canonical path** — since the canonical path contains no symlink segments at
* all, this eliminates check-then-use TOCTOU escapes (an out-of-bounds symlink
* is already resolved and rejected at the realpath step).
*/
private async resolveRead(workspace: string, rel: string): Promise<string> {
const realBase = await this.realBase(workspace);
const target = this.lexicalTarget(path.resolve(workspace), rel);
let canonical: string;
try {
canonical = await fs.realpath(target);
} catch (err) {
if ((err as NodeJS.ErrnoException).code === "ENOENT") {
throw new HttpError(404, "path_not_found", "文件不存在。");
}
throw err;
}
this.assertInside(canonical, realBase);
return canonical;
}
/**
* Write-path resolution: realpaths the parent directory (whose canonical path
* has no symlink segments) and checks containment, then appends the final
* segment as the file name. When the parent directory is missing, it is safely
* created (uploading a folder needs to preserve directory structure): first the
* nearest **existing** ancestor is found and its canonical path checked against
* the Workspace — this exposes it if a middle segment was preset as a symlink
* pointing outside; the missing segments are then created recursively beneath it
* (a brand-new directory can never be a symlink), followed by a second realpath
* check after creation. The actual write opens with O_NOFOLLOW (refusing to
* follow a symlink at the final segment), blocking the sandbox-escape pattern of
* "Agent presets a symlink -> an upload is used as leverage to overwrite a file
* outside the sandbox". Returns the canonical parent directory + file name.
*/
private async resolveWriteParent(
workspace: string,
rel: string,
): Promise<{ dir: string; name: string }> {
const realBase = await this.realBase(workspace);
const target = this.lexicalTarget(path.resolve(workspace), rel);
const name = path.basename(target);
if (name === "" || name === "." || name === "..") throw badRequest("path 必须是文件路径。");
const parent = path.dirname(target);
let canonicalParent: string;
try {
canonicalParent = await fs.realpath(parent);
} catch (err) {
if ((err as NodeJS.ErrnoException).code !== "ENOENT") throw err;
let probe = parent;
while (true) {
try {
this.assertInside(await fs.realpath(probe), realBase);
break;
} catch (probeErr) {
if ((probeErr as NodeJS.ErrnoException).code !== "ENOENT") throw probeErr;
const up = path.dirname(probe);
if (up === probe) throw badRequest("path 非法。");
probe = up;
}
}
await fs.mkdir(parent, { recursive: true });
canonicalParent = await fs.realpath(parent);
}
this.assertInside(canonicalParent, realBase);
return { dir: canonicalParent, name };
}
/**
* Batch existence check (a message's file card lists only files that actually
* exist): each item goes through the same containment resolution as reading
* (resolveRead); out-of-bounds, resolution failure, missing Workspace, or an
* irregular file are all treated as non-existent — the card scenario only asks
* "can this be opened", and throwing a 4xx would only add frontend branches while
* leaking containment details. Returns the deduplicated existing items in input order.
*/
async statExisting(workspace: string, rels: string[]): Promise<string[]> {
const unique = [...new Set(rels)];
const exists = await Promise.all(
unique.map(async (rel) => {
try {
const stat = await fs.stat(await this.resolveRead(workspace, rel));
return stat.isFile();
} catch {
return false;
}
}),
);
return unique.filter((_, i) => exists[i]);
}
/** List a directory: dirs come first, each group sorted by name; kind follows the symlink target (consistent with read behavior). */
async list(workspace: string, rel: string): Promise<WorkspaceFilesResponse> {
const dir = await this.resolveRead(workspace, rel);
let dirents;
try {
dirents = await fs.readdir(dir, { withFileTypes: true });
} catch (err) {
if ((err as NodeJS.ErrnoException).code === "ENOENT") {
throw new HttpError(404, "path_not_found", "目录不存在。");
}
if ((err as NodeJS.ErrnoException).code === "ENOTDIR") {
throw badRequest("path 不是目录。");
}
throw err;
}
const entries = await Promise.all(
dirents.map(async (d) => {
let sizeBytes = 0;
let mtime = "";
// Dirent doesn't report the target type for a symlink, so stat (following the link) is used to determine dir/file.
let isDir = d.isDirectory();
try {
const stat = await fs.stat(path.join(dir, d.name));
sizeBytes = stat.size;
mtime = stat.mtime.toISOString();
isDir = stat.isDirectory();
} catch {
// A dangling symlink or similar: keep the entry, with size/time left at defaults.
}
return {
name: d.name,
kind: isDir ? ("dir" as const) : ("file" as const),
sizeBytes,
mtime,
};
}),
);
entries.sort((a, b) =>
a.kind === b.kind ? a.name.localeCompare(b.name) : a.kind === "dir" ? -1 : 1,
);
return { path: rel, entries };
}
/** Read a file (preview/download): IO on the canonical path (resolveRead has already eliminated symlink escapes). */
async read(workspace: string, rel: string): Promise<WorkspaceFileContent> {
const file = await this.resolveRead(workspace, rel);
let stat;
try {
stat = await fs.stat(file);
} catch {
throw new HttpError(404, "path_not_found", "文件不存在。");
}
if (stat.isDirectory()) throw badRequest("path 是目录。");
if (stat.size > MAX_READ_BYTES) {
throw new HttpError(413, "file_too_large", "文件超过 50MB 读取上限。");
}
const data = await fs.readFile(file);
const ext = path.extname(file).toLowerCase();
return {
data,
fileName: path.basename(file),
contentType: CONTENT_TYPES[ext] ?? "application/octet-stream",
scriptable: ext === ".html" || ext === ".htm" || ext === ".svg",
};
}
/**
* Write a file (upload, overwriting a same-named one). If the parent directory
* is missing, it's automatically created under sandbox checks (preserving
* directory structure for folder uploads); the final segment is opened with
* O_NOFOLLOW, refusing to follow a symlink to write outside the Workspace
* (together with resolveWriteParent's canonical-parent check, this blocks
* sandbox escapes).
*/
async write(workspace: string, rel: string, data: Buffer): Promise<void> {
if (rel === "" || rel.endsWith("/")) throw badRequest("path 必须是文件路径。");
if (data.length > MAX_UPLOAD_BYTES) {
throw new HttpError(413, "file_too_large", "上传文件超过 14MB 上限。");
}
const { dir, name } = await this.resolveWriteParent(workspace, rel);
const file = path.join(dir, name);
// O_NOFOLLOW: open reports ELOOP if the final segment is a symlink, refusing to use it as leverage to overwrite a file outside the sandbox.
const flags = fsc.O_WRONLY | fsc.O_CREAT | fsc.O_TRUNC | (fsc.O_NOFOLLOW ?? 0);
let handle;
try {
handle = await fs.open(file, flags, 0o644);
} catch (err) {
const code = (err as NodeJS.ErrnoException).code;
if (code === "ELOOP") throw badRequest("path 不能是符号链接。");
if (code === "ENOENT") throw new HttpError(404, "path_not_found", "父目录不存在。");
if (code === "EISDIR") throw badRequest("path 是目录。");
throw err;
}
try {
await handle.writeFile(data);
} finally {
await handle.close();
}
}
}
@@ -0,0 +1,37 @@
/**
* Workspace validation.
*
* When a user explicitly specifies a Workspace: after realpath normalization
* (resolving `..` and symlinks), it's required to be an **existing directory**
* (never auto-created). The directory location is not constrained to the
* Project directory — a Workspace can be any path on the server, with actual
* reachability governed by the file permissions of the OS account running the
* service.
* When no Workspace is specified, this module isn't involved (the SDK creates its
* own temporary directory).
*/
import fs from "node:fs/promises";
import { HttpError } from "../http/errors.js";
/**
* Validates and returns the normalized (realpath) Workspace path.
*
* @throws 400 workspace_not_found: the path doesn't exist, isn't readable, or isn't a directory.
*/
export async function assertWorkspaceAllowed(args: { workspace: string }): Promise<string> {
let ws: string;
try {
ws = await fs.realpath(args.workspace);
} catch {
throw new HttpError(
400,
"workspace_not_found",
`Workspace 不存在或不可访问:${args.workspace}。请指定一个已存在的目录,或留空以使用临时目录。`,
);
}
const stat = await fs.stat(ws);
if (!stat.isDirectory()) {
throw new HttpError(400, "workspace_not_found", `Workspace 不是目录:${args.workspace}。`);
}
return ws;
}