Changelog, dev startup, README, AgentHub 0.4.0, model catalog, and landing site (#7)

Branch-length batch covering tooling, the model layer, the Web App and the public
surfaces. Highlights:

- Changelog: a per-release `changelog/<version>/` tree, grouped by the surface each
  change touches, with a root CHANGELOG.md holding one line per release.
- Dev startup: `scripts/dev-prebuild.mjs` serializes the skills+core prebuild behind a
  lock and keeps `pnpm install` current; `pnpm dev` runs server+web together.
- AgentHub 0.3.3 -> 0.4.0: OmniMessage complete payloads carry one opaque `fidelity`
  object in place of item-level `signature`/`phase`, threaded verbatim through Trace,
  replay and resume; malformed classification adapted to the new error types.
- Model layer: a model is always referenced by an explicit `(provider, model_id)` pair.
  The provider is never inferred, guessed or defaulted -- both the catalog inference and
  the unique-match config resolution are gone, and CLI, SDK, server routes and
  run_subagent all require the complete pair. Catalog gains the Qwen Token Plan, Qwen
  Pay-As-You-Go and Fireworks AI gateways, plus an expanded OpenRouter group.
- Web App: catalog preset sync and per-group speed test on the Models page, positional
  slash commands, a markdown renderer, skill-library update reminders, and a vertically
  centred draft page whose upward menus size themselves to the room available.
- Public surfaces: restructured READMEs, the penguin.ooo landing site and blog, refreshed
  benchmark results for both suites, and the demo videos playing on the landing page.

Includes the fixes from a full review of the branch: 23 confirmed findings, among them a
provider-inference bug that could send one vendor's API key to another vendor's endpoint,
and an Escape handler that destroyed the composer's contents unrecoverably.

Verified on the branch head: pnpm test (1127 passing, 7 packages), pnpm typecheck and
pnpm format:check clean, Playwright e2e 14/14.
This commit is contained in:
Yaowei Zheng
2026-07-21 17:43:31 +08:00
committed by GitHub
parent abf0a2f248
commit d4faee3a1e
214 changed files with 9406 additions and 1688 deletions
+34 -1
View File
@@ -10,6 +10,8 @@
* template's comments).
*/
import fs from "node:fs/promises";
import type { Dirent } from "node:fs";
import path from "node:path";
import { HttpError } from "../http/errors.js";
import {
agentDir,
@@ -20,6 +22,7 @@ import {
isValidId,
loadAgentVault,
scheduleDir,
skillsDir,
systemConfigPath,
} from "@prismshadow/penguin-core";
import type { AgentsRepo } from "../db/repos/agents.js";
@@ -41,6 +44,8 @@ export interface AgentListItem {
vaultKeyCount: number;
/** Number of scheduled tasks (count of .toml files under schedule/, including invalid ones). */
scheduleCount: number;
/** Number of installed Skills (count of skills/<name>/ directories that contain a SKILL.md). */
skillCount: number;
}
export class AgentService {
@@ -86,11 +91,12 @@ export class AgentService {
);
return Promise.all(
sorted.map(async (row) => {
const [meta, updatedAt, vaultKeyCount, scheduleCount] = await Promise.all([
const [meta, updatedAt, vaultKeyCount, scheduleCount, skillCount] = await Promise.all([
this.agentConfig.readCardMeta(projectId, row.agentId),
this.configUpdatedAt(projectId, row.agentId),
this.vaultKeyCount(projectId, row.agentId),
this.scheduleCount(projectId, row.agentId),
this.skillCount(projectId, row.agentId),
]);
return {
agentId: row.agentId,
@@ -99,6 +105,7 @@ export class AgentService {
...(updatedAt !== undefined ? { updatedAt } : {}),
vaultKeyCount,
scheduleCount,
skillCount,
};
}),
);
@@ -123,6 +130,30 @@ export class AgentService {
}
}
/** Number of installed Skills: count of skills/<name>/ directories containing a SKILL.md (0 if the directory doesn't exist). */
private async skillCount(projectId: string, agentId: string): Promise<number> {
const base = skillsDir(this.root, projectId, agentId);
let dirents: Dirent[];
try {
dirents = await fs.readdir(base, { withFileTypes: true });
} catch {
return 0;
}
const present = await Promise.all(
dirents
.filter((d) => d.isDirectory())
.map(async (d) => {
try {
await fs.access(path.join(base, d.name, "SKILL.md"));
return true;
} catch {
return false;
}
}),
);
return present.filter(Boolean).length;
}
/** Last config modification time: the later of system_config.yaml and AGENTS.md mtime; omitted if neither is readable. */
private async configUpdatedAt(projectId: string, agentId: string): Promise<string | undefined> {
const paths = [
@@ -218,6 +249,8 @@ export class AgentService {
version: meta.version,
vaultKeyCount: 0,
scheduleCount: 0,
// Read the real count: coreCreateAgent seeds the default skill set for default_agent.
skillCount: await this.skillCount(projectId, agentId),
};
}
}
@@ -27,7 +27,7 @@ import {
resolveModelEnv,
userText,
} from "@prismshadow/penguin-core";
import type { ModelRef } from "@prismshadow/penguin-core";
import type { LLMOutcome, ModelRef, OmniMessage } from "@prismshadow/penguin-core";
import type {
ModelInfo,
ModelPricingDto,
@@ -94,6 +94,26 @@ function showRef(provider: string, modelId: string): string {
return `(provider=${provider}, model_id=${modelId})`;
}
/**
* Connectivity probe prompt: asks for one word, so the whole exchange fits in a
* single-digit output budget. The wording discourages reasoning and the trailing empty
* <think></think> makes many reasoning models treat their thinking phase as already
* closed - keeping the probe's tiny budget on actual output instead of burning it on
* thinking.
*/
const PROBE_PROMPT =
'ping - reply with the single word "pong" and nothing else. Do not think or explain.\n<think></think>';
/**
* Speed probe prompt: a one-word answer can't be timed (a compliant model emits 1-3
* tokens, and the window is then dominated by the final usage chunk's round trip), so
* speed mode asks for something long enough to run into its raised cap. Counting to 50 is
* deterministic and needs no knowledge, so every model produces the same token stream at
* its own decoding rate. Carries the same anti-thinking hint as the connectivity prompt.
*/
const SPEED_PROBE_PROMPT =
"Count from 1 to 50 as a comma-separated list, and nothing else. Do not think or explain.\n<think></think>";
export class ProjectConfigService {
constructor(private readonly root: string) {}
@@ -203,12 +223,19 @@ export class ProjectConfigService {
* as a pair in the request body; sends one minimal request using that model's
* config (optionally overridden with an unsaved apiKey / baseUrl) — no tools, no
* system prompt, thinking disabled, a tiny output cap, 20s timeout — just to see
* whether it completes normally. The model id sent to AgentHub is `modelId`
* whether the endpoint answers. The model id sent to AgentHub is `modelId`
* itself (the upstream id verbatim; client_type inference follows it).
*
* A reasoning-heavy model may ignore the disabled thinking level and burn the
* whole tiny output cap on thinking (finish_reason=length with no text — AgentHub
* raises EmptyResponseError, collapsed to a malformed outcome): the endpoint
* demonstrably streamed model output, which is everything a connectivity test
* proves, so that case counts as ok too (see probeVerdict).
*
* Never throws: the LLM layer collapses auth/parameter/network errors into an
* `LLMOutcome`, which is translated here into ok / message. Consumes very few
* Tokens (single-digit output), and writes no Trace and records no usage.
* Tokens (single-digit output; speed mode spends up to its 64-token cap to have a
* window worth timing), and writes no Trace and records no usage.
*/
async testModel(projectId: string, req: ModelTestRequest): Promise<ModelTestResponse> {
const raw = await this.readRaw(projectId);
@@ -236,18 +263,43 @@ export class ProjectConfigService {
...(clientType ? { clientType } : {}),
tools: [],
thinkingLevel: "none",
maxTokens: 16,
// Speed mode pairs a raised cap with a prompt that keeps generating (see
// SPEED_PROBE_PROMPT), so the stream lasts long enough for TTFT/TPS to describe
// decoding rather than one round trip; the plain connectivity test keeps the
// single-digit-token budget.
maxTokens: req.speed ? 64 : 16,
requestTimeoutMs: 20_000,
});
const gen = llm.streamGenerate({ newMessages: [userText("ping")] });
const gen = llm.streamGenerate({
newMessages: [userText(req.speed ? SPEED_PROBE_PROMPT : PROBE_PROMPT)],
});
let sawContent = false;
let firstContentAt: number | null = null;
let outputTokens = 0;
for (;;) {
const step = await gen.next();
if (step.done) {
const outcome = step.value;
if (outcome.status === "completed")
return { ok: true, latencyMs: Date.now() - startedAt };
const detail = "message" in outcome && outcome.message ? outcome.message : outcome.status;
return { ok: false, message: String(detail).slice(0, 300) };
const verdict = probeVerdict(step.value, sawContent);
if (!verdict.ok) return verdict;
const res: ModelTestResponse = { ok: true, latencyMs: Date.now() - startedAt };
if (firstContentAt !== null) {
res.ttftMs = firstContentAt - startedAt;
// Output rate over the streaming window (first content -> stream end), dropped
// when the sample is too small to mean anything (see probeTps): usage is only
// reported on completed streams, so thinking-only malformed endings carry TTFT
// but no rate, and so does a model that answers in a couple of tokens.
const tps = probeTps(outputTokens, Date.now() - firstContentAt);
if (tps !== undefined) res.tps = tps;
}
return res;
}
if (isProbeContent(step.value)) {
sawContent = true;
if (firstContentAt === null) firstContentAt = Date.now();
}
const p = step.value.payload as { type?: string; request?: { output?: number } };
if (p.type === "token_usage" && typeof p.request?.output === "number") {
outputTokens = p.request.output;
}
}
} catch (err) {
@@ -496,3 +548,56 @@ export class ProjectConfigService {
return this.getModels(projectId);
}
}
/**
* Whether a streamed message carries genuine model content (thinking or text, partial delta
* or complete backfill) — the probe's "the endpoint really answered" signal. Tool calls and
* event messages don't count (the probe declares no tools).
*/
export function isProbeContent(msg: OmniMessage): boolean {
const p = msg.payload as { type?: string; thinking?: string; text?: string };
if (p.type === "partial_thinking" || p.type === "thinking") return Boolean(p.thinking);
if (p.type === "partial_text" || p.type === "text") return Boolean(p.text);
return false;
}
/**
* Probe verdict from the terminal LLM outcome. `completed` always passes. A `malformed`
* ending after genuine streamed content also passes: the typical case is a reasoning-heavy
* model that ignores the disabled thinking level and burns the probe's tiny max_tokens
* entirely on thinking (finish_reason=length -> AgentHub's EmptyResponseError) — the
* endpoint, credential, and model id all demonstrably work, which is what a connectivity
* test measures. Everything else (auth/parameter failures, timeouts, malformed with nothing
* received) fails with the outcome's message.
*/
export function probeVerdict(
outcome: LLMOutcome,
sawContent: boolean,
): { ok: true } | { ok: false; message: string } {
if (outcome.status === "completed") return { ok: true };
if (outcome.status === "malformed" && sawContent) return { ok: true };
const detail = "message" in outcome && outcome.message ? outcome.message : outcome.status;
return { ok: false, message: String(detail).slice(0, 300) };
}
/**
* Sample floors below which the streaming window says nothing about decoding rate. The
* speed probe's 64-token cap clears both by a wide margin, so hitting a floor means the
* model didn't really stream (a one-word answer, or usage that never arrived).
*/
const PROBE_TPS_MIN_TOKENS = 16;
const PROBE_TPS_MIN_WINDOW_MS = 100;
/**
* Output rate (tokens/s) over the probe's streaming window (first content -> stream end),
* rounded to 1dp; undefined when the sample is too small to be meaningful. A stream's
* closing usage chunk costs a round trip on its own, so a two-token answer measures network
* jitter and nothing else — 2 tokens in 30ms reads as 66.7 tok/s and the same model 30ms
* later reads as 33.3, which the card badges would paint green vs yellow. Callers report
* TTFT alone rather than a fabricated rate; a malformed ending carries no usage at all and
* lands here as 0 tokens.
*/
export function probeTps(outputTokens: number, windowMs: number): number | undefined {
if (outputTokens < PROBE_TPS_MIN_TOKENS || windowMs <= PROBE_TPS_MIN_WINDOW_MS) return undefined;
return Math.round((outputTokens / (windowMs / 1000)) * 10) / 10;
}
+25 -14
View File
@@ -7,9 +7,9 @@
* (provider, model_id) / workspace, which is backfilled into a DB row
* (approval_mode defaults, createdAt is taken from the timestamp embedded in
* session_id).
* Create: via core's `agent.createSession` (model reference as a provider + modelId
* pair; defaults to the Project's default reference, 400 if there is none; omitting
* provider goes through resolveModelRef for unique resolution); the new Session is
* Create: via core's `agent.createSession` (the model reference is always a complete
* (provider, modelId) pair — both or neither; omitting both falls back to the
* Project's default reference, 400 if there is none); the new Session is
* added to session-manager's active table (state idle).
*/
import path from "node:path";
@@ -22,6 +22,7 @@ import {
} from "@prismshadow/penguin-core";
import type { ApprovalMode, SessionInfo } from "../api/types.js";
import { HttpError, isMissingCredential, modelCredentialMissing } from "../http/errors.js";
import { badRequest } from "../http/validate.js";
import type { SessionRow, SessionsRepo } from "../db/repos/sessions.js";
import type { SessionManager } from "../runtime/session-manager.js";
import type { ProjectConfigService } from "./project-config-service.js";
@@ -141,28 +142,38 @@ export class SessionService {
}
/**
* Create a Session: model reference `(provider, modelId)` as a pair; defaults to
* the Project's default reference (400 prompting to configure a model first if
* there is none); when provider is omitted, core's resolveModelRef performs
* unique resolution (400 on zero matches / ambiguity). `workspace` is already
* Create a Session: the model reference is a complete `(provider, modelId)` pair.
* Half a reference is a client error, never something to resolve — the missing half
* is never guessed, since a guessed provider would send an entry's credential to a
* vendor nobody named. Omitting both falls back to the Project's default reference
* (400 prompting to configure a model first if there is none). `workspace` is already
* validated by the route guard. The new Session is added to the active table
* (idle).
*/
async createSession(args: {
projectId: string;
agentId: string;
/** Upstream id of the session's model (paired with provider); defaults to the Project's default reference. */
/** Upstream id of the session's model; always paired with provider. Omit both for the Project's default reference. */
modelId?: string;
/** The provider group for `modelId`; if omitted, resolveModelRef performs unique resolution. */
/** The provider group for `modelId`; always paired with modelId, never inferred. */
provider?: string;
workspace?: string;
approvalMode?: ApprovalMode;
/** Session source marker (schedule when triggered by a scheduled task; defaults to user-created). */
source?: "schedule";
}): Promise<SessionInfo> {
let modelId = args.modelId;
let provider = args.provider;
if (modelId === undefined) {
if ((args.modelId === undefined) !== (args.provider === undefined)) {
throw badRequest(
"modelId and provider must be given together as a (provider, modelId) pair: specify both, or neither to use the Project's default model.",
);
}
let modelId: string;
let provider: string;
if (args.modelId !== undefined && args.provider !== undefined) {
modelId = args.modelId;
provider = args.provider;
} else {
// The guard above leaves only "both omitted" here: fall back to the Project default.
const def = await this.deps.projectConfig.getDefaultModelRef(args.projectId);
if (def === undefined) {
throw new HttpError(
@@ -183,12 +194,12 @@ export class SessionService {
try {
session = await agent.createSession({
modelId,
...(provider !== undefined ? { provider } : {}),
provider,
...(args.workspace !== undefined ? { workspaceDir: args.workspace } : {}),
});
} catch (err) {
// A missing credential is its own category (the frontend shows localized text
// by code); other core errors (zero matches / ambiguous reference, Workspace
// by code); other core errors (the pair naming no configured entry, Workspace
// not existing, etc.) are collapsed to 400 — the guard already blocks most cases.
if (isMissingCredential(err)) throw modelCredentialMissing(modelId);
throw new HttpError(