Changelog, dev startup, README, AgentHub 0.4.0, model catalog, and landing site (#7)
Branch-length batch covering tooling, the model layer, the Web App and the public surfaces. Highlights: - Changelog: a per-release `changelog/<version>/` tree, grouped by the surface each change touches, with a root CHANGELOG.md holding one line per release. - Dev startup: `scripts/dev-prebuild.mjs` serializes the skills+core prebuild behind a lock and keeps `pnpm install` current; `pnpm dev` runs server+web together. - AgentHub 0.3.3 -> 0.4.0: OmniMessage complete payloads carry one opaque `fidelity` object in place of item-level `signature`/`phase`, threaded verbatim through Trace, replay and resume; malformed classification adapted to the new error types. - Model layer: a model is always referenced by an explicit `(provider, model_id)` pair. The provider is never inferred, guessed or defaulted -- both the catalog inference and the unique-match config resolution are gone, and CLI, SDK, server routes and run_subagent all require the complete pair. Catalog gains the Qwen Token Plan, Qwen Pay-As-You-Go and Fireworks AI gateways, plus an expanded OpenRouter group. - Web App: catalog preset sync and per-group speed test on the Models page, positional slash commands, a markdown renderer, skill-library update reminders, and a vertically centred draft page whose upward menus size themselves to the room available. - Public surfaces: restructured READMEs, the penguin.ooo landing site and blog, refreshed benchmark results for both suites, and the demo videos playing on the landing page. Includes the fixes from a full review of the branch: 23 confirmed findings, among them a provider-inference bug that could send one vendor's API key to another vendor's endpoint, and an Escape handler that destroyed the composer's contents unrecoverably. Verified on the branch head: pnpm test (1127 passing, 7 packages), pnpm typecheck and pnpm format:check clean, Playwright e2e 14/14.
This commit is contained in:
@@ -10,6 +10,8 @@
|
||||
* template's comments).
|
||||
*/
|
||||
import fs from "node:fs/promises";
|
||||
import type { Dirent } from "node:fs";
|
||||
import path from "node:path";
|
||||
import { HttpError } from "../http/errors.js";
|
||||
import {
|
||||
agentDir,
|
||||
@@ -20,6 +22,7 @@ import {
|
||||
isValidId,
|
||||
loadAgentVault,
|
||||
scheduleDir,
|
||||
skillsDir,
|
||||
systemConfigPath,
|
||||
} from "@prismshadow/penguin-core";
|
||||
import type { AgentsRepo } from "../db/repos/agents.js";
|
||||
@@ -41,6 +44,8 @@ export interface AgentListItem {
|
||||
vaultKeyCount: number;
|
||||
/** Number of scheduled tasks (count of .toml files under schedule/, including invalid ones). */
|
||||
scheduleCount: number;
|
||||
/** Number of installed Skills (count of skills/<name>/ directories that contain a SKILL.md). */
|
||||
skillCount: number;
|
||||
}
|
||||
|
||||
export class AgentService {
|
||||
@@ -86,11 +91,12 @@ export class AgentService {
|
||||
);
|
||||
return Promise.all(
|
||||
sorted.map(async (row) => {
|
||||
const [meta, updatedAt, vaultKeyCount, scheduleCount] = await Promise.all([
|
||||
const [meta, updatedAt, vaultKeyCount, scheduleCount, skillCount] = await Promise.all([
|
||||
this.agentConfig.readCardMeta(projectId, row.agentId),
|
||||
this.configUpdatedAt(projectId, row.agentId),
|
||||
this.vaultKeyCount(projectId, row.agentId),
|
||||
this.scheduleCount(projectId, row.agentId),
|
||||
this.skillCount(projectId, row.agentId),
|
||||
]);
|
||||
return {
|
||||
agentId: row.agentId,
|
||||
@@ -99,6 +105,7 @@ export class AgentService {
|
||||
...(updatedAt !== undefined ? { updatedAt } : {}),
|
||||
vaultKeyCount,
|
||||
scheduleCount,
|
||||
skillCount,
|
||||
};
|
||||
}),
|
||||
);
|
||||
@@ -123,6 +130,30 @@ export class AgentService {
|
||||
}
|
||||
}
|
||||
|
||||
/** Number of installed Skills: count of skills/<name>/ directories containing a SKILL.md (0 if the directory doesn't exist). */
|
||||
private async skillCount(projectId: string, agentId: string): Promise<number> {
|
||||
const base = skillsDir(this.root, projectId, agentId);
|
||||
let dirents: Dirent[];
|
||||
try {
|
||||
dirents = await fs.readdir(base, { withFileTypes: true });
|
||||
} catch {
|
||||
return 0;
|
||||
}
|
||||
const present = await Promise.all(
|
||||
dirents
|
||||
.filter((d) => d.isDirectory())
|
||||
.map(async (d) => {
|
||||
try {
|
||||
await fs.access(path.join(base, d.name, "SKILL.md"));
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}),
|
||||
);
|
||||
return present.filter(Boolean).length;
|
||||
}
|
||||
|
||||
/** Last config modification time: the later of system_config.yaml and AGENTS.md mtime; omitted if neither is readable. */
|
||||
private async configUpdatedAt(projectId: string, agentId: string): Promise<string | undefined> {
|
||||
const paths = [
|
||||
@@ -218,6 +249,8 @@ export class AgentService {
|
||||
version: meta.version,
|
||||
vaultKeyCount: 0,
|
||||
scheduleCount: 0,
|
||||
// Read the real count: coreCreateAgent seeds the default skill set for default_agent.
|
||||
skillCount: await this.skillCount(projectId, agentId),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +27,7 @@ import {
|
||||
resolveModelEnv,
|
||||
userText,
|
||||
} from "@prismshadow/penguin-core";
|
||||
import type { ModelRef } from "@prismshadow/penguin-core";
|
||||
import type { LLMOutcome, ModelRef, OmniMessage } from "@prismshadow/penguin-core";
|
||||
import type {
|
||||
ModelInfo,
|
||||
ModelPricingDto,
|
||||
@@ -94,6 +94,26 @@ function showRef(provider: string, modelId: string): string {
|
||||
return `(provider=${provider}, model_id=${modelId})`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Connectivity probe prompt: asks for one word, so the whole exchange fits in a
|
||||
* single-digit output budget. The wording discourages reasoning and the trailing empty
|
||||
* <think></think> makes many reasoning models treat their thinking phase as already
|
||||
* closed - keeping the probe's tiny budget on actual output instead of burning it on
|
||||
* thinking.
|
||||
*/
|
||||
const PROBE_PROMPT =
|
||||
'ping - reply with the single word "pong" and nothing else. Do not think or explain.\n<think></think>';
|
||||
|
||||
/**
|
||||
* Speed probe prompt: a one-word answer can't be timed (a compliant model emits 1-3
|
||||
* tokens, and the window is then dominated by the final usage chunk's round trip), so
|
||||
* speed mode asks for something long enough to run into its raised cap. Counting to 50 is
|
||||
* deterministic and needs no knowledge, so every model produces the same token stream at
|
||||
* its own decoding rate. Carries the same anti-thinking hint as the connectivity prompt.
|
||||
*/
|
||||
const SPEED_PROBE_PROMPT =
|
||||
"Count from 1 to 50 as a comma-separated list, and nothing else. Do not think or explain.\n<think></think>";
|
||||
|
||||
export class ProjectConfigService {
|
||||
constructor(private readonly root: string) {}
|
||||
|
||||
@@ -203,12 +223,19 @@ export class ProjectConfigService {
|
||||
* as a pair in the request body; sends one minimal request using that model's
|
||||
* config (optionally overridden with an unsaved apiKey / baseUrl) — no tools, no
|
||||
* system prompt, thinking disabled, a tiny output cap, 20s timeout — just to see
|
||||
* whether it completes normally. The model id sent to AgentHub is `modelId`
|
||||
* whether the endpoint answers. The model id sent to AgentHub is `modelId`
|
||||
* itself (the upstream id verbatim; client_type inference follows it).
|
||||
*
|
||||
* A reasoning-heavy model may ignore the disabled thinking level and burn the
|
||||
* whole tiny output cap on thinking (finish_reason=length with no text — AgentHub
|
||||
* raises EmptyResponseError, collapsed to a malformed outcome): the endpoint
|
||||
* demonstrably streamed model output, which is everything a connectivity test
|
||||
* proves, so that case counts as ok too (see probeVerdict).
|
||||
*
|
||||
* Never throws: the LLM layer collapses auth/parameter/network errors into an
|
||||
* `LLMOutcome`, which is translated here into ok / message. Consumes very few
|
||||
* Tokens (single-digit output), and writes no Trace and records no usage.
|
||||
* Tokens (single-digit output; speed mode spends up to its 64-token cap to have a
|
||||
* window worth timing), and writes no Trace and records no usage.
|
||||
*/
|
||||
async testModel(projectId: string, req: ModelTestRequest): Promise<ModelTestResponse> {
|
||||
const raw = await this.readRaw(projectId);
|
||||
@@ -236,18 +263,43 @@ export class ProjectConfigService {
|
||||
...(clientType ? { clientType } : {}),
|
||||
tools: [],
|
||||
thinkingLevel: "none",
|
||||
maxTokens: 16,
|
||||
// Speed mode pairs a raised cap with a prompt that keeps generating (see
|
||||
// SPEED_PROBE_PROMPT), so the stream lasts long enough for TTFT/TPS to describe
|
||||
// decoding rather than one round trip; the plain connectivity test keeps the
|
||||
// single-digit-token budget.
|
||||
maxTokens: req.speed ? 64 : 16,
|
||||
requestTimeoutMs: 20_000,
|
||||
});
|
||||
const gen = llm.streamGenerate({ newMessages: [userText("ping")] });
|
||||
const gen = llm.streamGenerate({
|
||||
newMessages: [userText(req.speed ? SPEED_PROBE_PROMPT : PROBE_PROMPT)],
|
||||
});
|
||||
let sawContent = false;
|
||||
let firstContentAt: number | null = null;
|
||||
let outputTokens = 0;
|
||||
for (;;) {
|
||||
const step = await gen.next();
|
||||
if (step.done) {
|
||||
const outcome = step.value;
|
||||
if (outcome.status === "completed")
|
||||
return { ok: true, latencyMs: Date.now() - startedAt };
|
||||
const detail = "message" in outcome && outcome.message ? outcome.message : outcome.status;
|
||||
return { ok: false, message: String(detail).slice(0, 300) };
|
||||
const verdict = probeVerdict(step.value, sawContent);
|
||||
if (!verdict.ok) return verdict;
|
||||
const res: ModelTestResponse = { ok: true, latencyMs: Date.now() - startedAt };
|
||||
if (firstContentAt !== null) {
|
||||
res.ttftMs = firstContentAt - startedAt;
|
||||
// Output rate over the streaming window (first content -> stream end), dropped
|
||||
// when the sample is too small to mean anything (see probeTps): usage is only
|
||||
// reported on completed streams, so thinking-only malformed endings carry TTFT
|
||||
// but no rate, and so does a model that answers in a couple of tokens.
|
||||
const tps = probeTps(outputTokens, Date.now() - firstContentAt);
|
||||
if (tps !== undefined) res.tps = tps;
|
||||
}
|
||||
return res;
|
||||
}
|
||||
if (isProbeContent(step.value)) {
|
||||
sawContent = true;
|
||||
if (firstContentAt === null) firstContentAt = Date.now();
|
||||
}
|
||||
const p = step.value.payload as { type?: string; request?: { output?: number } };
|
||||
if (p.type === "token_usage" && typeof p.request?.output === "number") {
|
||||
outputTokens = p.request.output;
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
@@ -496,3 +548,56 @@ export class ProjectConfigService {
|
||||
return this.getModels(projectId);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a streamed message carries genuine model content (thinking or text, partial delta
|
||||
* or complete backfill) — the probe's "the endpoint really answered" signal. Tool calls and
|
||||
* event messages don't count (the probe declares no tools).
|
||||
*/
|
||||
export function isProbeContent(msg: OmniMessage): boolean {
|
||||
const p = msg.payload as { type?: string; thinking?: string; text?: string };
|
||||
if (p.type === "partial_thinking" || p.type === "thinking") return Boolean(p.thinking);
|
||||
if (p.type === "partial_text" || p.type === "text") return Boolean(p.text);
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Probe verdict from the terminal LLM outcome. `completed` always passes. A `malformed`
|
||||
* ending after genuine streamed content also passes: the typical case is a reasoning-heavy
|
||||
* model that ignores the disabled thinking level and burns the probe's tiny max_tokens
|
||||
* entirely on thinking (finish_reason=length -> AgentHub's EmptyResponseError) — the
|
||||
* endpoint, credential, and model id all demonstrably work, which is what a connectivity
|
||||
* test measures. Everything else (auth/parameter failures, timeouts, malformed with nothing
|
||||
* received) fails with the outcome's message.
|
||||
*/
|
||||
export function probeVerdict(
|
||||
outcome: LLMOutcome,
|
||||
sawContent: boolean,
|
||||
): { ok: true } | { ok: false; message: string } {
|
||||
if (outcome.status === "completed") return { ok: true };
|
||||
if (outcome.status === "malformed" && sawContent) return { ok: true };
|
||||
const detail = "message" in outcome && outcome.message ? outcome.message : outcome.status;
|
||||
return { ok: false, message: String(detail).slice(0, 300) };
|
||||
}
|
||||
|
||||
/**
|
||||
* Sample floors below which the streaming window says nothing about decoding rate. The
|
||||
* speed probe's 64-token cap clears both by a wide margin, so hitting a floor means the
|
||||
* model didn't really stream (a one-word answer, or usage that never arrived).
|
||||
*/
|
||||
const PROBE_TPS_MIN_TOKENS = 16;
|
||||
const PROBE_TPS_MIN_WINDOW_MS = 100;
|
||||
|
||||
/**
|
||||
* Output rate (tokens/s) over the probe's streaming window (first content -> stream end),
|
||||
* rounded to 1dp; undefined when the sample is too small to be meaningful. A stream's
|
||||
* closing usage chunk costs a round trip on its own, so a two-token answer measures network
|
||||
* jitter and nothing else — 2 tokens in 30ms reads as 66.7 tok/s and the same model 30ms
|
||||
* later reads as 33.3, which the card badges would paint green vs yellow. Callers report
|
||||
* TTFT alone rather than a fabricated rate; a malformed ending carries no usage at all and
|
||||
* lands here as 0 tokens.
|
||||
*/
|
||||
export function probeTps(outputTokens: number, windowMs: number): number | undefined {
|
||||
if (outputTokens < PROBE_TPS_MIN_TOKENS || windowMs <= PROBE_TPS_MIN_WINDOW_MS) return undefined;
|
||||
return Math.round((outputTokens / (windowMs / 1000)) * 10) / 10;
|
||||
}
|
||||
|
||||
@@ -7,9 +7,9 @@
|
||||
* (provider, model_id) / workspace, which is backfilled into a DB row
|
||||
* (approval_mode defaults, createdAt is taken from the timestamp embedded in
|
||||
* session_id).
|
||||
* Create: via core's `agent.createSession` (model reference as a provider + modelId
|
||||
* pair; defaults to the Project's default reference, 400 if there is none; omitting
|
||||
* provider goes through resolveModelRef for unique resolution); the new Session is
|
||||
* Create: via core's `agent.createSession` (the model reference is always a complete
|
||||
* (provider, modelId) pair — both or neither; omitting both falls back to the
|
||||
* Project's default reference, 400 if there is none); the new Session is
|
||||
* added to session-manager's active table (state idle).
|
||||
*/
|
||||
import path from "node:path";
|
||||
@@ -22,6 +22,7 @@ import {
|
||||
} from "@prismshadow/penguin-core";
|
||||
import type { ApprovalMode, SessionInfo } from "../api/types.js";
|
||||
import { HttpError, isMissingCredential, modelCredentialMissing } from "../http/errors.js";
|
||||
import { badRequest } from "../http/validate.js";
|
||||
import type { SessionRow, SessionsRepo } from "../db/repos/sessions.js";
|
||||
import type { SessionManager } from "../runtime/session-manager.js";
|
||||
import type { ProjectConfigService } from "./project-config-service.js";
|
||||
@@ -141,28 +142,38 @@ export class SessionService {
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a Session: model reference `(provider, modelId)` as a pair; defaults to
|
||||
* the Project's default reference (400 prompting to configure a model first if
|
||||
* there is none); when provider is omitted, core's resolveModelRef performs
|
||||
* unique resolution (400 on zero matches / ambiguity). `workspace` is already
|
||||
* Create a Session: the model reference is a complete `(provider, modelId)` pair.
|
||||
* Half a reference is a client error, never something to resolve — the missing half
|
||||
* is never guessed, since a guessed provider would send an entry's credential to a
|
||||
* vendor nobody named. Omitting both falls back to the Project's default reference
|
||||
* (400 prompting to configure a model first if there is none). `workspace` is already
|
||||
* validated by the route guard. The new Session is added to the active table
|
||||
* (idle).
|
||||
*/
|
||||
async createSession(args: {
|
||||
projectId: string;
|
||||
agentId: string;
|
||||
/** Upstream id of the session's model (paired with provider); defaults to the Project's default reference. */
|
||||
/** Upstream id of the session's model; always paired with provider. Omit both for the Project's default reference. */
|
||||
modelId?: string;
|
||||
/** The provider group for `modelId`; if omitted, resolveModelRef performs unique resolution. */
|
||||
/** The provider group for `modelId`; always paired with modelId, never inferred. */
|
||||
provider?: string;
|
||||
workspace?: string;
|
||||
approvalMode?: ApprovalMode;
|
||||
/** Session source marker (schedule when triggered by a scheduled task; defaults to user-created). */
|
||||
source?: "schedule";
|
||||
}): Promise<SessionInfo> {
|
||||
let modelId = args.modelId;
|
||||
let provider = args.provider;
|
||||
if (modelId === undefined) {
|
||||
if ((args.modelId === undefined) !== (args.provider === undefined)) {
|
||||
throw badRequest(
|
||||
"modelId and provider must be given together as a (provider, modelId) pair: specify both, or neither to use the Project's default model.",
|
||||
);
|
||||
}
|
||||
let modelId: string;
|
||||
let provider: string;
|
||||
if (args.modelId !== undefined && args.provider !== undefined) {
|
||||
modelId = args.modelId;
|
||||
provider = args.provider;
|
||||
} else {
|
||||
// The guard above leaves only "both omitted" here: fall back to the Project default.
|
||||
const def = await this.deps.projectConfig.getDefaultModelRef(args.projectId);
|
||||
if (def === undefined) {
|
||||
throw new HttpError(
|
||||
@@ -183,12 +194,12 @@ export class SessionService {
|
||||
try {
|
||||
session = await agent.createSession({
|
||||
modelId,
|
||||
...(provider !== undefined ? { provider } : {}),
|
||||
provider,
|
||||
...(args.workspace !== undefined ? { workspaceDir: args.workspace } : {}),
|
||||
});
|
||||
} catch (err) {
|
||||
// A missing credential is its own category (the frontend shows localized text
|
||||
// by code); other core errors (zero matches / ambiguous reference, Workspace
|
||||
// by code); other core errors (the pair naming no configured entry, Workspace
|
||||
// not existing, etc.) are collapsed to 400 — the guard already blocks most cases.
|
||||
if (isMissingCredential(err)) throw modelCredentialMissing(modelId);
|
||||
throw new HttpError(
|
||||
|
||||
Reference in New Issue
Block a user