Changelog, dev startup, README, AgentHub 0.4.0, model catalog, and landing site (#7)
Branch-length batch covering tooling, the model layer, the Web App and the public surfaces. Highlights: - Changelog: a per-release `changelog/<version>/` tree, grouped by the surface each change touches, with a root CHANGELOG.md holding one line per release. - Dev startup: `scripts/dev-prebuild.mjs` serializes the skills+core prebuild behind a lock and keeps `pnpm install` current; `pnpm dev` runs server+web together. - AgentHub 0.3.3 -> 0.4.0: OmniMessage complete payloads carry one opaque `fidelity` object in place of item-level `signature`/`phase`, threaded verbatim through Trace, replay and resume; malformed classification adapted to the new error types. - Model layer: a model is always referenced by an explicit `(provider, model_id)` pair. The provider is never inferred, guessed or defaulted -- both the catalog inference and the unique-match config resolution are gone, and CLI, SDK, server routes and run_subagent all require the complete pair. Catalog gains the Qwen Token Plan, Qwen Pay-As-You-Go and Fireworks AI gateways, plus an expanded OpenRouter group. - Web App: catalog preset sync and per-group speed test on the Models page, positional slash commands, a markdown renderer, skill-library update reminders, and a vertically centred draft page whose upward menus size themselves to the room available. - Public surfaces: restructured READMEs, the penguin.ooo landing site and blog, refreshed benchmark results for both suites, and the demo videos playing on the landing page. Includes the fixes from a full review of the branch: 23 confirmed findings, among them a provider-inference bug that could send one vendor's API key to another vendor's endpoint, and an Escape handler that destroyed the composer's contents unrecoverably. Verified on the branch head: pnpm test (1127 passing, 7 packages), pnpm typecheck and pnpm format:check clean, Playwright e2e 14/14.
This commit is contained in:
@@ -1,10 +1,10 @@
|
||||
/**
|
||||
* Integration tests for `penguin config model add|default|vision|list` (run through
|
||||
* commander's parseAsync for the full command path): --model-id always takes the
|
||||
* upstream id, paired with --provider to form a (provider, model_id) reference (add's
|
||||
* --provider defaults to catalog-based inference, falling back to custom when
|
||||
* inference fails; default / vision require --provider and raise an error when the
|
||||
* reference isn't found in models — no string concatenation is ever performed); --root
|
||||
* upstream id, paired with --provider to form a (provider, model_id) reference
|
||||
* (--provider is required on all three subcommands — the group is never inferred — and
|
||||
* default / vision raise an error when the reference isn't found in models; no string
|
||||
* concatenation is ever performed); --root
|
||||
* specifies the data root directory (takes priority over PENGUIN_HOME); persisted to a
|
||||
* single hidden .project_config.toml (mode 0600, credentials inline, provider and
|
||||
* model_id as separate columns); list displays provider and model_id as separate
|
||||
@@ -84,13 +84,15 @@ describe("penguin config model add/list (--root plus provider / model_id stored
|
||||
"add",
|
||||
"--model-id",
|
||||
"my-own-model",
|
||||
"--provider",
|
||||
"custom",
|
||||
"--api-key",
|
||||
"sk-root-secret-1",
|
||||
"--root",
|
||||
tmpRoot,
|
||||
]);
|
||||
expect(add.code).toBe(0);
|
||||
// Catalog inference fails -> falls back to the custom group (provider is a separate field, never concatenated into the id).
|
||||
// The named group is stored as a separate field, never concatenated into the id.
|
||||
expect(add.out).toContain("Added model (provider=custom, model_id=my-own-model).");
|
||||
|
||||
const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID);
|
||||
@@ -119,11 +121,13 @@ describe("penguin config model add/list (--root plus provider / model_id stored
|
||||
expect(list.out).not.toContain("sk-root-secret-1");
|
||||
});
|
||||
|
||||
it("built-in catalog infers the grouping: an upstream id matching the catalog lands under that provider; --set-default writes a pair reference", async () => {
|
||||
it("naming an existing pair updates that preset entry in place; --set-default writes a pair reference", async () => {
|
||||
const add = await runModel([
|
||||
"add",
|
||||
"--model-id",
|
||||
"claude-sonnet-4-6",
|
||||
"--provider",
|
||||
"anthropic",
|
||||
"--set-default",
|
||||
"--root",
|
||||
tmpRoot,
|
||||
@@ -168,8 +172,16 @@ describe("penguin config model add/list (--root plus provider / model_id stored
|
||||
});
|
||||
|
||||
it("client_type defaults by grouping semantics (PRN-021): custom / self-hosted / gateway get openai, first-party providers get none", async () => {
|
||||
// custom (catalog inference fails) and self-hosted groups (--provider not a catalog value): default to client_type=openai.
|
||||
await runModel(["add", "--model-id", "my-openai-proxy", "--root", tmpRoot]);
|
||||
// The custom group and self-hosted groups (--provider not a catalog value): default to client_type=openai.
|
||||
await runModel([
|
||||
"add",
|
||||
"--model-id",
|
||||
"my-openai-proxy",
|
||||
"--provider",
|
||||
"custom",
|
||||
"--root",
|
||||
tmpRoot,
|
||||
]);
|
||||
await runModel(["add", "--model-id", "in-house-1", "--provider", "mylab", "--root", tmpRoot]);
|
||||
// A non-catalog id under a first-party vendor group: client_type is not set (AgentHub auto-routes by upstream id).
|
||||
await runModel([
|
||||
@@ -241,13 +253,29 @@ describe("penguin config model add/list (--root plus provider / model_id stored
|
||||
});
|
||||
});
|
||||
|
||||
describe("model default/vision: --provider is required, (provider, model_id) pair reference", () => {
|
||||
describe("model add/default/vision: --provider is required, (provider, model_id) pair reference", () => {
|
||||
it("missing --provider: commander usage error, nonzero exit code", async () => {
|
||||
const bad = await runModel(["default", "--model-id", "deepseek-v4-flash", "--root", tmpRoot]);
|
||||
expect(bad.code).not.toBe(0);
|
||||
expect(bad.err).toContain("--provider");
|
||||
});
|
||||
|
||||
it("add without --provider is a usage error too: the group is never inferred, so no config is written", async () => {
|
||||
const bad = await runModel([
|
||||
"add",
|
||||
"--model-id",
|
||||
"claude-sonnet-4-6",
|
||||
"--api-key",
|
||||
"sk-never-stored",
|
||||
"--root",
|
||||
tmpRoot,
|
||||
]);
|
||||
expect(bad.code).not.toBe(0);
|
||||
expect(bad.err).toContain("--provider");
|
||||
// The credential must not have landed on a guessed vendor: nothing was persisted at all.
|
||||
await expect(fs.access(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID))).rejects.toThrow();
|
||||
});
|
||||
|
||||
it("dangling reference: the pair is not in models; the error carries the pair reference and a model list hint", async () => {
|
||||
const bad = await runModel([
|
||||
"default",
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
/**
|
||||
* `penguin run` / `penguin chat`: a model reference is always an explicit
|
||||
* (provider, model_id) pair. Commander can only mark each option required on its own, so
|
||||
* the "both or neither" rule is enforced inside the action — supplying exactly one of
|
||||
* --model-id / --provider is a usage error (never a lookup against the configured models),
|
||||
* while supplying neither falls back to the Project's default model.
|
||||
*/
|
||||
import fs from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { Command } from "commander";
|
||||
import { registerChatCommand } from "../src/commands/chat.js";
|
||||
import { registerRunCommand } from "../src/commands/run.js";
|
||||
import { getMessages } from "../src/i18n.js";
|
||||
|
||||
// The --resume case below reaches createAgent, which initializes Agent state on disk; point
|
||||
// the data root at a throwaway directory so no test ever writes to the real ~/.penguin/data.
|
||||
let tmpHome: string;
|
||||
let prevHome: string | undefined;
|
||||
|
||||
beforeEach(async () => {
|
||||
prevHome = process.env.PENGUIN_HOME;
|
||||
tmpHome = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-pairing-"));
|
||||
process.env.PENGUIN_HOME = tmpHome;
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
if (prevHome === undefined) delete process.env.PENGUIN_HOME;
|
||||
else process.env.PENGUIN_HOME = prevHome;
|
||||
await fs.rm(tmpHome, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
/** Runs one command, capturing stdout / stderr and the exit code (without exiting the process). */
|
||||
async function runCommand(
|
||||
register: (program: Command, t: ReturnType<typeof getMessages>) => void,
|
||||
args: string[],
|
||||
): Promise<{ out: string; err: string; code: number }> {
|
||||
const program = new Command();
|
||||
program.exitOverride();
|
||||
register(program, getMessages("en"));
|
||||
const out: string[] = [];
|
||||
const err: string[] = [];
|
||||
const outSpy = vi.spyOn(process.stdout, "write").mockImplementation((chunk) => {
|
||||
out.push(String(chunk));
|
||||
return true;
|
||||
});
|
||||
const errSpy = vi.spyOn(process.stderr, "write").mockImplementation((chunk) => {
|
||||
err.push(String(chunk));
|
||||
return true;
|
||||
});
|
||||
const prevExitCode = process.exitCode;
|
||||
process.exitCode = undefined;
|
||||
try {
|
||||
await program.parseAsync(["node", "penguin", ...args]);
|
||||
return { out: out.join(""), err: err.join(""), code: Number(process.exitCode ?? 0) };
|
||||
} catch (e) {
|
||||
const exitCode = (e as { exitCode?: number }).exitCode;
|
||||
return { out: out.join(""), err: err.join(""), code: exitCode || 1 };
|
||||
} finally {
|
||||
outSpy.mockRestore();
|
||||
errSpy.mockRestore();
|
||||
process.exitCode = prevExitCode;
|
||||
}
|
||||
}
|
||||
|
||||
describe("run: --model-id and --provider must be given together", () => {
|
||||
it("--model-id without --provider: error on stderr, exit code 1", async () => {
|
||||
const bad = await runCommand(registerRunCommand, [
|
||||
"run",
|
||||
"-m",
|
||||
"hi",
|
||||
"--model-id",
|
||||
"deepseek-v4-flash",
|
||||
]);
|
||||
expect(bad.code).toBe(1);
|
||||
expect(bad.err).toContain("--model-id and --provider must be given together");
|
||||
});
|
||||
|
||||
it("--provider without --model-id: same error (the pair is never half-specified)", async () => {
|
||||
const bad = await runCommand(registerRunCommand, ["run", "-m", "hi", "--provider", "deepseek"]);
|
||||
expect(bad.code).toBe(1);
|
||||
expect(bad.err).toContain("--model-id and --provider must be given together");
|
||||
});
|
||||
});
|
||||
|
||||
describe("chat: --model-id and --provider must be given together", () => {
|
||||
it("--model-id without --provider: error on stderr, exit code 1", async () => {
|
||||
const bad = await runCommand(registerChatCommand, ["chat", "--model-id", "deepseek-v4-flash"]);
|
||||
expect(bad.code).toBe(1);
|
||||
expect(bad.err).toContain("--model-id and --provider must be given together");
|
||||
});
|
||||
|
||||
it("--resume plus a lone --model-id keeps the more specific resume error", async () => {
|
||||
const bad = await runCommand(registerChatCommand, [
|
||||
"chat",
|
||||
"--resume",
|
||||
"sess-1",
|
||||
"--model-id",
|
||||
"deepseek-v4-flash",
|
||||
]);
|
||||
expect(bad.code).toBe(1);
|
||||
expect(bad.out).toContain("--resume does not accept");
|
||||
expect(`${bad.out}${bad.err}`).not.toContain("must be given together");
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user