diff --git a/CHANGELOG.md b/CHANGELOG.md index e95e09f..854de64 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,5 +2,6 @@ One brief line per release. Per-release detail lives in [`changelog//`](changelog/). +- **0.1.1** — 2026-07-22. Gemini 3.6 Flash and 3.5 Flash-Lite in the model catalog, a Workspace-grouped chat sidebar, in-place upgrades via `penguin update`, and the vLLM / Ollama / LlamaFactory and bento-slides skills. ([details](changelog/0.1.1/README.md)) - **0.1.0** — 2026-07-21. First feature release: the Web App, landing site and docs, the model catalog, and the self-improvement skills. ([details](changelog/0.1.0/README.md)) - **0.0.1** — 2026-07-19. First tagged release; changelog history starts after this tag. diff --git a/README.md b/README.md index 63aa7ed..e6df5de 100644 --- a/README.md +++ b/README.md @@ -147,6 +147,7 @@ for await (const output of session.run([userText("Create hello.txt containing hi - [ ] Windows support - [ ] Agent company and templates - [ ] Company-level self evolving +- [ ] OpenShell integration (permission-governed shell) - More to come… ## Development diff --git a/README.zh.md b/README.zh.md index c27d072..530643d 100644 --- a/README.zh.md +++ b/README.zh.md @@ -147,6 +147,7 @@ for await (const output of session.run([userText("Create hello.txt containing hi - [ ] Windows 系统支持 - [ ] Agent 公司与模板 - [ ] 公司级自进化能力 +- [ ] 集成 OpenShell(带权限管控的 shell) - 更多规划,敬请期待…… ## 参与开发 diff --git a/changelog/0.2.0/2026-07-22-docs-and-examples.md b/changelog/0.1.1/2026-07-22-docs-and-examples.md similarity index 100% rename from changelog/0.2.0/2026-07-22-docs-and-examples.md rename to changelog/0.1.1/2026-07-22-docs-and-examples.md diff --git a/changelog/0.2.0/2026-07-22-models-and-core.md b/changelog/0.1.1/2026-07-22-models-and-core.md similarity index 100% rename from changelog/0.2.0/2026-07-22-models-and-core.md rename to changelog/0.1.1/2026-07-22-models-and-core.md diff --git a/changelog/0.2.0/2026-07-22-sites-and-blog.md b/changelog/0.1.1/2026-07-22-sites-and-blog.md similarity index 100% rename from changelog/0.2.0/2026-07-22-sites-and-blog.md rename to changelog/0.1.1/2026-07-22-sites-and-blog.md diff --git a/changelog/0.2.0/2026-07-22-skills.md b/changelog/0.1.1/2026-07-22-skills.md similarity index 63% rename from changelog/0.2.0/2026-07-22-skills.md rename to changelog/0.1.1/2026-07-22-skills.md index 217248a..43af848 100644 --- a/changelog/0.2.0/2026-07-22-skills.md +++ b/changelog/0.1.1/2026-07-22-skills.md @@ -1,4 +1,4 @@ -# Skills: local serving and fine-tuning join AI App Development +# Skills: local serving and fine-tuning join AI App Development, plus deck authoring Three new built-in skills extend the AI App Development group so agents can stand up and tune the models they build on: @@ -10,4 +10,6 @@ Both serving skills share a guided workflow: ask the user which model to serve ( The hard root-separation guardrail lives where configuration is taught — the `penguin-cli` skill (now v5) and the `penguin-sdk` skill: configuring Penguin's own model uses the default root, but models configured for an AI app under development must use the app's own project directory (`--root ./penguin_data`) unless the user chose otherwise — never the global `~/.penguin/data`. The serving skills point at that rule rather than restating it. +**bento-slides** joins the Office Productivity group, so an agent asked for a presentation produces a real deck rather than a wall of bullets. A Bento deck is one self-contained `.bento.html` file whose document is plain JSON in a `#bento-doc` script block; the skill teaches the agent to edit that block in place, to fetch the app itself from bento.page when starting from nothing, and — the part that matters — to map source material onto the right feature (charts for numbers, tables for grids, morph transitions for a subject that changes across slides, state slides for drill-downs, ken-burns and count-ups for motion) instead of defaulting to text slides. It carries the gotchas that otherwise cost a round trip: chart series data must be plain numbers, morph needs stable shared element ids, assets are embedded as data URIs, and `docId` is never regenerated on an existing deck. Adapted from the Bento project's own skill (MIT, © 2026 The Bento/Suite authors) with attribution in the skill body; the authoritative schema stays at https://bento.page/agents.md. + The `agenthub-models` skill tracks the AgentHub 0.4.1 upgrade: it documents the new supported-model registry, the config parameters a client may now reject outright, and the Gemini 3.6 / Kimi K3 / GLM-5.2 families with their reasoning-effort knobs, alongside a routing table rewritten from the release's actual client-matching order. diff --git a/changelog/0.1.1/2026-07-22-tooling.md b/changelog/0.1.1/2026-07-22-tooling.md new file mode 100644 index 0000000..534187a --- /dev/null +++ b/changelog/0.1.1/2026-07-22-tooling.md @@ -0,0 +1,5 @@ +# Tooling and hardening: in-place upgrades, request validation and core test coverage + +- **`penguin update` upgrades an existing install in place.** `penguin update [--check] [--release ] [-y|--yes]` resolves the newest version from the GitHub Releases API (the same source `install.sh` resolves `releases/latest/download` against) and upgrades using the mechanism the install actually came from, detected from the real path of the running CLI rather than guessed: a tarball install re-runs the official installer preserving its install dir and whether it bundles a Node runtime; a global npm/pnpm/yarn/bun install runs that manager's global install, and prints the command instead of guessing when the manager cannot be identified; a source checkout is refused with a pointer to `git pull` and a rebuild. Without `-y` it prints the mechanism, target version and install dir and asks for confirmation, and a non-TTY stdin requires `--yes` rather than blocking. The data root is never touched — only `bin`, `lib`, `web` and `node` are replaced. The target flag is `--release`, not `--version`, because the CLI's own `-v, --version` takes precedence over a subcommand option of the same name. +- The server now validates `positiveIntParam` and `optionalDateParam` inputs instead of trusting query strings — malformed paging/date parameters return a clean 400 rather than leaking into SQL or arithmetic. +- Core gained dedicated unit tests for `CappedTextBuffer` and `ToolCallIdAllocator`, two small pure modules that previously had no direct coverage. diff --git a/changelog/0.2.0/2026-07-22-web-app.md b/changelog/0.1.1/2026-07-22-web-app.md similarity index 100% rename from changelog/0.2.0/2026-07-22-web-app.md rename to changelog/0.1.1/2026-07-22-web-app.md diff --git a/changelog/0.2.0/README.md b/changelog/0.1.1/README.md similarity index 82% rename from changelog/0.2.0/README.md rename to changelog/0.1.1/README.md index f9646e5..d2ed193 100644 --- a/changelog/0.2.0/README.md +++ b/changelog/0.1.1/README.md @@ -1,15 +1,15 @@ -# Version 0.2.0 +# Version 0.1.1 -Unreleased. +Released on 2026-07-22. - [2026-07-22] Models and core: empty tool lists are omitted from LLM requests (fixing 400s from strict OpenAI-compatible servers), the default system prompt gains service-protection and API-key retry guardrails with the default port as a core SDK constant, a per-model max output tokens cap lands on the Models page, the thinking level moves to a conversation-time picker (low and above) that writes through to Agent settings, subagents inherit the parent session's model and thinking level, sessions record their origin in `session_meta.source` as the single source of truth, the SDK moves to AgentHub 0.4.1 and its supported-model registry drives a catalog refresh across every provider group (with the READMEs trimmed to the newest generation per vendor), and session-title generation folds into core's internal module. ([details](2026-07-22-models-and-core.md)) - [2026-07-22] Web App: the chat sidebar groups conversations by Workspace (with an Agent-mode toggle, group pinning, subagent/scheduled folders and paged loading), the collapsed sidebar becomes an eight-entry navigation rail with bilingual tooltips, the model picker lists key-configured models first, chat renders links in a new tab with clean CJK/URL wrapping and the subagent expansion below the tool's own output, mobile dropdowns stay inside the viewport, the Cost center's daily-token tooltip follows the pointer and shows the cache hit rate, the model and Agent settings forms were tightened, the copied task-stats line is localized, and custom model groups and Agents get initial-letter avatars. ([details](2026-07-22-web-app.md)) -- [2026-07-22] Skills: vLLM and Ollama deployment plus LlamaFactory fine-tuning join the AI App Development group with a guided serving workflow that follows the user's engine preference, and `agenthub-models` tracks the AgentHub 0.4.1 API. ([details](2026-07-22-skills.md)) +- [2026-07-22] Skills: vLLM and Ollama deployment plus LlamaFactory fine-tuning join the AI App Development group with a guided serving workflow that follows the user's engine preference, `bento-slides` joins Office Productivity for authoring single-file Bento decks, and `agenthub-models` tracks the AgentHub 0.4.1 API. ([details](2026-07-22-skills.md)) - [2026-07-22] Sites: the docs and landing navbars are now identical, the blog gains Tech-practice and Perspectives categories, pinned posts, author/date/copy-link metadata, a second AMD practice post and three bilingual Perspectives posts (harness minimalism against the Databricks benchmark, a sourced comparison of five agent frameworks, and the AI development stack — vLLM, Ollama, LlamaFactory — as infrastructure now driven by agents rather than people), and the built-in Skills are listed in the READMEs and on the landing page. ([details](2026-07-22-sites-and-blog.md)) - [2026-07-22] Docs and examples: two new README roadmap items, and the self-improvement example reworked to genuinely evolve itself. ([details](2026-07-22-docs-and-examples.md)) -- [2026-07-22] Tooling: server query-parameter validation hardening, and unit tests for two previously uncovered core modules. ([details](2026-07-22-tooling.md)) +- [2026-07-22] Tooling: `penguin update` upgrades an existing install in place via the mechanism it was installed with, plus server query-parameter validation hardening and unit tests for two previously uncovered core modules. ([details](2026-07-22-tooling.md)) diff --git a/changelog/0.2.0/2026-07-22-tooling.md b/changelog/0.2.0/2026-07-22-tooling.md deleted file mode 100644 index 499e389..0000000 --- a/changelog/0.2.0/2026-07-22-tooling.md +++ /dev/null @@ -1,4 +0,0 @@ -# Tooling and hardening: request validation and core test coverage - -- The server now validates `positiveIntParam` and `optionalDateParam` inputs instead of trusting query strings — malformed paging/date parameters return a clean 400 rather than leaking into SQL or arithmetic. -- Core gained dedicated unit tests for `CappedTextBuffer` and `ToolCallIdAllocator`, two small pure modules that previously had no direct coverage. diff --git a/package.json b/package.json index 8bec10c..d0ccfa0 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "penguin-harness", - "version": "0.1.0", + "version": "0.1.1", "private": true, "type": "module", "description": "PenguinHarness — TypeScript AI Agent (SDK + CLI).", diff --git a/packages/cli/package.json b/packages/cli/package.json index 44f1a8b..0dcc07f 100644 --- a/packages/cli/package.json +++ b/packages/cli/package.json @@ -1,6 +1,6 @@ { "name": "@prismshadow/penguin-cli", - "version": "0.1.0", + "version": "0.1.1", "type": "module", "description": "PenguinHarness CLI: interactive REPL and single-task runner over @prismshadow/penguin-core.", "license": "Apache-2.0", diff --git a/packages/cli/src/commands/update.ts b/packages/cli/src/commands/update.ts new file mode 100644 index 0000000..16c8b7a --- /dev/null +++ b/packages/cli/src/commands/update.ts @@ -0,0 +1,521 @@ +/** + * `penguin update` — upgrades an existing install in place. + * + * penguin update [--check] [--release ] [-y|--yes] + * + * There is no single upgrade mechanism, because there is no single install mechanism: the + * documented path is the tarball installer (install.sh unpacks bin/lib/web/node into + * PENGUIN_INSTALL_DIR, default ~/.penguin), some users have a global npm install of + * @prismshadow/penguin-cli, and developers run out of a source checkout. This command works out + * which one it is from the real path of the running CLI and upgrades the way that install was + * made — never by guessing. A source checkout is refused outright: overwriting a working tree + * would destroy uncommitted work. + * + * The latest version comes from the GitHub Releases API, the same source of truth install.sh + * resolves `releases/latest/download` against. The target-a-specific-release flag is spelled + * `--release ` rather than `--version `: commander's program-level `-v, --version` + * intercepts a subcommand's own `--version` when it is written with a space, so + * `penguin update --version 0.1.2` would silently print the CLI version and do nothing. A flag + * that works only in its `--version=0.1.2` form is a trap, so it got an unambiguous name. + * + * Self-replacement hazard, and how it is handled: for a tarball install the installer deletes and + * replaces `lib/`, which is the directory this very process is executing from. Two things make + * that safe. (1) The upgrade runs in a child `sh`, from a script written to a private temporary + * directory — not inside the tree being replaced — so the script itself is never pulled out from + * under its own interpreter. (2) The parent does nothing after the swap begins that would touch + * the replaced tree: every module it needs is statically imported at the top of the bundle and + * therefore fully loaded before any action runs, every message it will print is resolved up front, + * and after the child exits it only removes its own temp directory, writes already-computed + * strings and sets an exit code. It never `import()`s anything (the CLI's only dynamic import is + * `@prismshadow/penguin-server`, reachable solely from the serve commands), and never re-reads a + * file. On POSIX an unlinked file stays valid for whoever has it open, so the running process is + * unaffected. + * + * Where the installer script is written, and why it matters: `tmpdir()` is world-writable and + * shared, so a fixed or guessable name there is a local privilege-escalation primitive — another + * user can pre-create the path as a symlink and have this process overwrite a file it owns, or + * swap the file between the write and the `spawn`, turning the upgrade into arbitrary code + * execution as the invoking user. The script therefore goes into a fresh `mkdtempSync` directory: + * created 0700 with an unpredictable name, so no one else can name the path in advance, and the + * file is written with `wx` so an existing entry is an error rather than a silent truncation. The + * directory is removed only after the child `sh` has exited (see the note at the call site). + * + * Windows never reaches either upgrade path. The tarball installer is a POSIX shell script; and a + * global npm/pnpm/yarn/bun install cannot be driven from here either, because `spawn` without a + * shell does no PATHEXT resolution and Node has refused to exec `.cmd` shims without one since the + * CVE-2024-27980 fix. Rather than fall through to a generic failure — or spawn through `cmd.exe`, + * which would interpolate a user-supplied release tag into a command line — both cases are refused + * up front with the command the user should run themselves. + * + * The data root (~/.penguin/data) is never touched — the installer only replaces bin/lib/web/node + * — and the confirmation prompt says so, because that is the thing users worry about. + * Docs: /docs/cli § "penguin update". + */ +import { spawn } from "node:child_process"; +import { createInterface } from "node:readline"; +import { existsSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { homedir, tmpdir } from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { realpathSync } from "node:fs"; +import { VERSION } from "@prismshadow/penguin-core"; +import type { Command } from "commander"; +import type { Messages } from "../i18n.js"; + +/** Repository the released artifacts come from — the same repo install.sh downloads from. */ +export const REPO_SLUG = "Prism-Shadow/penguin-harness"; +/** Releases API endpoint for the newest published release. */ +export const LATEST_RELEASE_API = `https://api.github.com/repos/${REPO_SLUG}/releases/latest`; + +/** How this copy of the CLI was installed, which decides how it can be upgraded. */ +export type InstallKind = "tarball" | "npm" | "source" | "unknown"; + +export interface InstallInfo { + kind: InstallKind; + /** Tarball only: the install dir that holds bin/lib/web/node (i.e. PENGUIN_INSTALL_DIR). */ + installDir?: string; + /** npm only: the global `node_modules` root that owns the package, used to identify the manager. */ + globalRoot?: string; +} + +/** Global-install package managers we can drive; `null` means "tell the user, do not guess". */ +export type PackageManager = "pnpm" | "npm" | "yarn" | "bun"; + +/** Splits a path into segments on either separator, so the same logic works for win32 paths. */ +function segments(p: string): string[] { + return p.split(/[\\/]+/).filter(Boolean); +} + +/** + * Works out how this CLI was installed from the resolved real path of its own module. + * + * Pure and shape-based on purpose: it touches neither the filesystem nor the environment beyond + * what is passed in, so every branch is unit-testable. The three layouts are unambiguous: + * + * - source checkout — `…/packages/cli/{src,dist}/index.{ts,js}` (also how a `pnpm link`-ed dev + * build looks once the bin symlink is resolved); + * - npm global — the path runs through `node_modules/@prismshadow/penguin-cli/`; + * - tarball — `/lib/dist/index.js`, the layout install.sh unpacks. + * + * Order matters: a checkout is checked first so a repo that happens to live under a directory + * called `lib` cannot be mistaken for an install. + */ +export function detectInstall(modulePath: string): InstallInfo { + const parts = segments(modulePath); + const sep = modulePath.includes("\\") && !modulePath.includes("/") ? "\\" : path.sep; + const join = (upto: number) => { + const joined = parts.slice(0, upto).join(sep); + return modulePath.startsWith("/") ? `${sep}${joined}` : joined; + }; + + // …/packages/cli/(src|dist)/index.(ts|js) + const cliIdx = parts.findIndex( + (seg, i) => seg === "packages" && parts[i + 1] === "cli" && i + 2 < parts.length, + ); + if (cliIdx >= 0 && (parts[cliIdx + 2] === "src" || parts[cliIdx + 2] === "dist")) { + return { kind: "source" }; + } + + // …/node_modules/@prismshadow/penguin-cli/… + const nmIdx = parts.findIndex( + (seg, i) => + seg === "node_modules" && parts[i + 1] === "@prismshadow" && parts[i + 2] === "penguin-cli", + ); + if (nmIdx >= 0) { + return { kind: "npm", globalRoot: join(nmIdx + 1) }; + } + + // /lib/dist/index.js + if (parts.length >= 3) { + const [lib, dist] = [parts[parts.length - 3], parts[parts.length - 2]]; + if (lib === "lib" && dist === "dist") { + return { kind: "tarball", installDir: join(parts.length - 3) }; + } + } + + return { kind: "unknown" }; +} + +/** + * Identifies the package manager that owns a global `node_modules` root, from the root's own path. + * Returns null when it is not recognizable — the caller then prints the command for the user to + * run rather than guessing, because an `npm i -g` over a pnpm-managed install leaves two copies + * and a broken shim. + */ +export function detectPackageManager(globalRoot: string): PackageManager | null { + const parts = segments(globalRoot).map((s) => s.toLowerCase()); + if (parts.includes(".pnpm") || (parts.includes("pnpm") && parts.includes("global"))) + return "pnpm"; + if (parts.includes(".bun")) return "bun"; + if (parts.includes("yarn") && parts.includes("global")) return "yarn"; + if (parts.includes("npm") || parts.includes("lib") || parts.includes("node_modules")) + return "npm"; + return null; +} + +/** The global-install command for a manager, at a specific version. */ +export function globalInstallCommand( + manager: PackageManager, + version: string, +): { command: string; args: string[] } { + const spec = `@prismshadow/penguin-cli@${version}`; + if (manager === "pnpm") return { command: "pnpm", args: ["add", "-g", spec] }; + if (manager === "yarn") return { command: "yarn", args: ["global", "add", spec] }; + if (manager === "bun") return { command: "bun", args: ["add", "-g", spec] }; + return { command: "npm", args: ["install", "-g", spec] }; +} + +/** Strips a leading `v` so `v0.1.2` and `0.1.2` are the same input. */ +export function normalizeVersion(tag: string): string { + return tag.trim().replace(/^v/i, ""); +} + +/** + * Compares two dotted numeric versions: -1 / 0 / 1. + * + * Each dot-separated component is read with `Number.parseInt`, which takes the leading digits and + * ignores the rest: `1abc` is 1, and `2-rc1` is 2. A component with no leading digit at all, or + * one that is missing entirely, counts as 0 — which is the property that matters, because it means + * a malformed or truncated tag can never make an upgrade look available. + * + * The consequence, stated rather than papered over: suffixes are invisible here, so `0.1.2-rc1` + * compares *equal* to `0.1.2`. This project tags plain `vX.Y.Z` releases only, and the API this + * reads (`tag_name` from GitHub Releases) returns those tags, so the case does not arise; carrying + * a full semver precedence implementation — with its own numeric-vs-alphanumeric identifier rules + * — to handle tags we do not publish would be more code and more ways to be wrong. If pre-release + * tags are ever published, this has to become a real semver compare before `--release` can target + * one. + */ +export function compareVersions(a: string, b: string): number { + const parse = (v: string) => + normalizeVersion(v) + .split(".") + .map((n) => Number.parseInt(n, 10)); + const [x, y] = [parse(a), parse(b)]; + for (let i = 0; i < Math.max(x.length, y.length); i += 1) { + const l = Number.isFinite(x[i]) ? (x[i] as number) : 0; + const r = Number.isFinite(y[i]) ? (y[i] as number) : 0; + if (l !== r) return l < r ? -1 : 1; + } + return 0; +} + +/** + * Builds the argv and environment for re-running install.sh, preserving the shape of the install + * being upgraded rather than the defaults: + * + * - `PENGUIN_INSTALL_DIR` is passed whenever the install is not at the default `~/.penguin`, or + * the upgrade would silently relocate it; + * - `--universal` is passed when the current install has no bundled `node/` directory, or the user + * would silently gain a runtime they deliberately did not install (and lose it in reverse); + * - `PENGUIN_VERSION` pins the target when `--release` was given. + * + * Pure so every combination is unit-testable; the caller supplies the two facts that need the + * filesystem (`installDir`, `hasBundledNode`). + */ +export function buildInstallerInvocation(opts: { + scriptPath: string; + installDir: string; + hasBundledNode: boolean; + defaultInstallDir: string; + version?: string; +}): { args: string[]; env: Record } { + const args = [opts.scriptPath]; + if (!opts.hasBundledNode) args.push("--universal"); + const env: Record = {}; + if (path.resolve(opts.installDir) !== path.resolve(opts.defaultInstallDir)) { + env.PENGUIN_INSTALL_DIR = opts.installDir; + } + if (opts.version) env.PENGUIN_VERSION = `v${normalizeVersion(opts.version)}`; + return { args, env }; +} + +/** Download URL for the installer of a given release (latest when no version is pinned). */ +export function installerUrl(version?: string): string { + const base = `https://github.com/${REPO_SLUG}/releases`; + return version + ? `${base}/download/v${normalizeVersion(version)}/install.sh` + : `${base}/latest/download/install.sh`; +} + +/** + * Resolves the newest published version from the Releases API. Every failure mode the API actually + * produces gets its own message instead of a stack trace: no network, a rate-limited 403 (which + * arrives with no useful body from an unauthenticated client), any other HTTP status, and a body + * that parses but carries no usable `tag_name`. + */ +export async function fetchLatestVersion(t: Messages): Promise { + let res: Response; + try { + res = await fetch(LATEST_RELEASE_API, { + headers: { accept: "application/vnd.github+json", "user-agent": "penguin-cli" }, + signal: AbortSignal.timeout(15_000), + }); + } catch { + throw new Error(t.update.networkFailed(LATEST_RELEASE_API)); + } + if (res.status === 403 || res.status === 429) throw new Error(t.update.rateLimited()); + if (!res.ok) throw new Error(t.update.apiFailed(res.status)); + let tag: unknown; + try { + tag = ((await res.json()) as { tag_name?: unknown }).tag_name; + } catch { + throw new Error(t.update.apiMalformed()); + } + if (typeof tag !== "string" || normalizeVersion(tag) === "") + throw new Error(t.update.apiMalformed()); + return normalizeVersion(tag); +} + +/** Interactive y/N confirmation; SIGINT and stream close both count as "no", so it can never hang. */ +function confirmYes(prompt: string): Promise { + const rl = createInterface({ input: process.stdin, output: process.stdout }); + return new Promise((resolve) => { + let done = false; + const finish = (value: boolean) => { + if (done) return; + done = true; + process.off("SIGINT", onSigint); + rl.close(); + resolve(value); + }; + const onSigint = () => finish(false); + process.once("SIGINT", onSigint); + rl.on("close", () => finish(false)); + rl.question(prompt, (answer) => finish(/^y(es)?$/i.test(answer.trim()))); + }); +} + +/** Runs a child process to completion, inheriting stdio; resolves with its exit code. */ +function run(command: string, args: string[], env: Record): Promise { + return new Promise((resolve) => { + const child = spawn(command, args, { + stdio: "inherit", + env: { ...process.env, ...env }, + }); + child.on("error", () => resolve(-1)); + child.on("close", (code) => resolve(code ?? -1)); + }); +} + +/** The real path of this module, with the ~/.local/bin/penguin symlink resolved. */ +function selfPath(): string { + const p = fileURLToPath(import.meta.url); + try { + return realpathSync(p); + } catch { + return p; + } +} + +/** + * What the command decided to do, once the two versions and the install layout are known. + * + * `report` and `up-to-date` change nothing; `refuse` prints a reason and stops; `npm` and + * `tarball` are the only two that go on to confirm and spawn anything. + */ +export type UpdatePlan = + | { action: "report"; current: string; target: string; comparison: number } + | { action: "up-to-date"; current: string } + | { action: "refuse"; reason: "source" } + | { action: "refuse"; reason: "unknown-install"; modulePath: string } + | { action: "refuse"; reason: "unknown-manager"; globalRoot: string; target: string } + | { action: "refuse"; reason: "windows-global"; command: string } + | { action: "refuse"; reason: "windows-installer" } + | { action: "npm"; manager: PackageManager; command: string; args: string[] } + | { action: "tarball"; installDir: string }; + +/** + * The whole decision, as one pure function: which of the outcomes above this invocation is, given + * the running version, the resolved target, the flags, and the layout this CLI was installed as. + * + * Everything that needs the outside world stays with the caller — resolving the latest version, + * probing for a bundled `node/`, prompting, spawning — so each branch below (including the two + * refusals that exist only to avoid a misleading failure) is directly testable without a network + * or a child process. The order is deliberate: `--check` reports and never upgrades, an install + * already on the target is done before its layout even matters, a source checkout and an + * unrecognised layout are refused before anything is downloaded, and Windows is refused inside + * whichever branch applies so its message can name the command that would have run. + */ +export function planUpdate(input: { + current: string; + target: string; + check?: boolean; + install: InstallInfo; + modulePath: string; + platform: string; + /** `~/.penguin`, passed in rather than read, so the tarball branch stays pure. */ + defaultInstallDir: string; +}): UpdatePlan { + const { current, target, install } = input; + const comparison = compareVersions(target, current); + + if (input.check) return { action: "report", current, target, comparison }; + if (comparison === 0) return { action: "up-to-date", current }; + + if (install.kind === "source") return { action: "refuse", reason: "source" }; + if (install.kind === "unknown") + return { action: "refuse", reason: "unknown-install", modulePath: input.modulePath }; + + if (install.kind === "npm") { + const globalRoot = install.globalRoot ?? ""; + const manager = detectPackageManager(globalRoot); + if (!manager) return { action: "refuse", reason: "unknown-manager", globalRoot, target }; + const { command, args } = globalInstallCommand(manager, target); + if (input.platform === "win32") + return { + action: "refuse", + reason: "windows-global", + command: `${command} ${args.join(" ")}`, + }; + return { action: "npm", manager, command, args }; + } + + if (input.platform === "win32") return { action: "refuse", reason: "windows-installer" }; + return { action: "tarball", installDir: install.installDir ?? input.defaultInstallDir }; +} + +/** The line printed for a refusal — one message per reason, no fallthrough. */ +function refusalMessage(plan: Extract, t: Messages): string { + switch (plan.reason) { + case "source": + return t.update.sourceCheckout(); + case "unknown-install": + return t.update.unknownInstall(plan.modulePath); + case "unknown-manager": + return t.update.npmUnknownManager(plan.globalRoot, plan.target); + case "windows-global": + return t.update.windowsGlobalInstall(plan.command); + case "windows-installer": + return t.update.windowsUnsupported(); + } +} + +export function registerUpdateCommand(program: Command, t: Messages): void { + program + .command("update") + .description(t.update.desc) + .option("--check", t.update.check) + .option("--release ", t.update.releaseOpt) + .option("-y, --yes", t.update.yes) + .action(async (opts: { check?: boolean; release?: string; yes?: boolean }) => { + const current = VERSION; + const target = opts.release ? normalizeVersion(opts.release) : await fetchLatestVersion(t); + const modulePath = selfPath(); + const defaultInstallDir = path.join(homedir(), ".penguin"); + const plan = planUpdate({ + current, + target, + ...(opts.check !== undefined ? { check: opts.check } : {}), + install: detectInstall(modulePath), + modulePath, + platform: process.platform, + defaultInstallDir, + }); + + if (plan.action === "report") { + process.stdout.write(`${t.update.checkReport(plan.current, plan.target)}\n`); + process.stdout.write( + `${plan.comparison > 0 ? t.update.upgradeAvailable(plan.target) : plan.comparison < 0 ? t.update.targetIsOlder(plan.target) : t.update.upToDate(plan.current)}\n`, + ); + return; + } + if (plan.action === "up-to-date") { + process.stdout.write(`${t.update.upToDate(plan.current)}\n`); + return; + } + if (plan.action === "refuse") { + process.stdout.write(`${refusalMessage(plan, t)}\n`); + return; + } + + // --- npm global install --- + if (plan.action === "npm") { + process.stdout.write( + `${t.update.planNpm(current, target, plan.manager, `${plan.command} ${plan.args.join(" ")}`)}\n`, + ); + if (!(await confirmUpgrade(opts.yes, t))) return; + const code = await run(plan.command, plan.args, {}); + process.stdout.write(`${code === 0 ? t.update.done(target) : t.update.failed()}\n`); + if (code !== 0) process.exitCode = 1; + return; + } + + // --- tarball install: re-run the installer, preserving this install's shape --- + const installDir = plan.installDir; + const hasBundledNode = existsSync(path.join(installDir, "node")); + process.stdout.write( + `${t.update.planTarball(current, target, installDir, !hasBundledNode)}\n`, + ); + if (!(await confirmUpgrade(opts.yes, t))) return; + + const url = installerUrl(opts.release ? target : undefined); + let script: string; + try { + const res = await fetch(url, { signal: AbortSignal.timeout(30_000) }); + if (!res.ok) throw new Error(String(res.status)); + script = await res.text(); + } catch { + process.stdout.write(`${t.update.installerFetchFailed(url)}\n`); + process.exitCode = 1; + return; + } + + // A private 0700 directory with an unpredictable name, never a fixed path in the shared + // world-writable temp dir: nobody else can pre-create this path as a symlink to overwrite, + // or swap the file out between the write below and the spawn. `wx` refuses to truncate an + // existing entry rather than inheriting its mode and owner. It is outside installDir on + // purpose — the installer replaces that tree, and a shell script deleted mid-execution is + // not something to rely on. + const scriptDir = mkdtempSync(path.join(tmpdir(), "penguin-update-")); + const scriptPath = path.join(scriptDir, "install.sh"); + writeFileSync(scriptPath, script, { mode: 0o700, flag: "wx" }); + + const { args, env } = buildInstallerInvocation({ + scriptPath, + installDir, + hasBundledNode, + defaultInstallDir, + version: opts.release ? target : undefined, + }); + // Past this point the installer may delete the tree this process runs from. Everything below + // is already-loaded code and already-resolved strings: no import, no file read, no re-entry. + const code = await run("sh", args, env); + // Safe only here: `close` has fired, so the child `sh` has exited and nothing is still + // reading the script (install.sh runs to completion synchronously and backgrounds nothing — + // it does its own `trap 'rm -rf "$TMP"' EXIT` cleanup and then returns). Deleting it any + // earlier would pull the script out from under a running interpreter. `force` keeps a + // failed cleanup from turning a successful upgrade into an error. + rmSync(scriptDir, { recursive: true, force: true }); + process.stdout.write(`${code === 0 ? t.update.done(target) : t.update.failed()}\n`); + if (code !== 0) process.exitCode = 1; + }); +} + +/** + * How the confirmation gate resolves, decided before any I/O: `--yes` proceeds outright, a + * non-interactive stdio pair has to be *told* rather than asked — a prompt nobody can answer would + * hang the upgrade forever in a pipe or a CI job — and anything else gets the interactive prompt. + * + * Pure, so the "must pass --yes" path is testable without a pseudo-terminal. + */ +export function confirmationMode( + yes: boolean | undefined, + interactive: boolean, +): "proceed" | "needs-yes" | "prompt" { + if (yes) return "proceed"; + return interactive ? "prompt" : "needs-yes"; +} + +/** Confirmation gate: `--yes` skips it; a non-TTY stdin must pass `--yes` rather than hang. */ +async function confirmUpgrade(yes: boolean | undefined, t: Messages): Promise { + const mode = confirmationMode(yes, Boolean(process.stdin.isTTY && process.stdout.isTTY)); + if (mode === "proceed") return true; + if (mode === "needs-yes") { + process.stdout.write(`${t.update.needsYes()}\n`); + return false; + } + if (await confirmYes(t.update.confirm())) return true; + process.stdout.write(`${t.update.cancelled()}\n`); + return false; +} diff --git a/packages/cli/src/i18n.ts b/packages/cli/src/i18n.ts index 8b24e05..fce84f4 100644 --- a/packages/cli/src/i18n.ts +++ b/packages/cli/src/i18n.ts @@ -72,6 +72,46 @@ export interface Messages { host: string; noOpen: string; }; + /** `penguin update`: help text and every line the command can print. */ + update: { + desc: string; + check: string; + releaseOpt: string; + yes: string; + /** `--check` header: the running version against the resolved target. */ + checkReport(current: string, latest: string): string; + upgradeAvailable(target: string): string; + upToDate(current: string): string; + /** `--release` naming a release older than the running one: allowed, but said out loud. */ + targetIsOlder(target: string): string; + /** Pre-confirmation plan for a tarball install (mechanism, target, install dir, data-dir guarantee). */ + planTarball(current: string, target: string, installDir: string, universal: boolean): string; + /** Pre-confirmation plan for a global npm install. */ + planNpm(current: string, target: string, manager: string, command: string): string; + confirm(): string; + /** stdin is not a TTY: require --yes instead of blocking on a prompt nobody can answer. */ + needsYes(): string; + cancelled(): string; + done(version: string): string; + failed(): string; + /** Running from a source checkout: refuse, because overwriting a working tree destroys work. */ + sourceCheckout(): string; + unknownInstall(modulePath: string): string; + /** A global install whose package manager could not be identified: print the command, never guess. */ + npmUnknownManager(globalRoot: string, target: string): string; + /** Windows, tarball install: the official installer is a POSIX shell script. */ + windowsUnsupported(): string; + /** + * Windows, global install: `spawn` cannot run a `.cmd` shim without a shell, so hand the user + * the exact command instead of failing generically. + */ + windowsGlobalInstall(command: string): string; + networkFailed(url: string): string; + rateLimited(): string; + apiFailed(status: number): string; + apiMalformed(): string; + installerFetchFailed(url: string): string; + }; // —— Runtime output —— header(kind: "chat" | "run", agentId: string, workspace: string, model: string): string; @@ -212,6 +252,61 @@ const en: Messages = { host: "Listen address (falls back to the HOST env var, default 127.0.0.1)", noOpen: "Do not open a browser automatically", }, + update: { + desc: "Upgrade this PenguinHarness install in place", + check: "Only report the current and latest versions; change nothing", + releaseOpt: + "Target a specific release tag instead of the latest (e.g. v0.1.2 or 0.1.2); named --release because -v/--version is the CLI's own version flag", + yes: "Skip the confirmation prompt", + checkReport: (current, latest) => `Installed ${current} · latest ${latest}`, + upgradeAvailable: (target) => + `An upgrade is available: run \`penguin update\` to install ${target}.`, + upToDate: (current) => `Already on the latest version (${current}); nothing to do.`, + targetIsOlder: (target) => + `${target} is older than the installed version — this would be a downgrade.`, + planTarball: (current, target, installDir, universal) => + [ + `Upgrade ${current} -> ${target}`, + ` how: re-run the official installer (this install came from the tarball)`, + ` install dir: ${installDir}${universal ? " (universal package, no bundled Node runtime)" : ""}`, + ` replaces: bin, lib, web${universal ? "" : ", node"} — your data dir is NOT touched`, + ].join("\n"), + planNpm: (current, target, manager, command) => + [ + `Upgrade ${current} -> ${target}`, + ` how: global ${manager} install (this install came from ${manager})`, + ` command: ${command}`, + ` your data dir is NOT touched`, + ].join("\n"), + confirm: () => "Proceed? [y/N] ", + needsYes: () => + "Not running in a terminal, so the confirmation cannot be answered. Re-run with --yes to upgrade non-interactively.", + cancelled: () => "Cancelled; nothing was changed.", + done: (version) => + `PenguinHarness ${version} installed. Run \`penguin --version\` in a new shell to confirm.`, + failed: () => "Upgrade failed; the previous install was left in place where possible.", + sourceCheckout: () => + "This penguin runs from a source checkout, so there is nothing to download — update it with `git pull` and rebuild (`pnpm install && pnpm -r build`).", + unknownInstall: (modulePath) => + `Cannot tell how this penguin was installed (running from ${modulePath}), so it will not be replaced. Re-install with the official installer, or upgrade with the package manager you used.`, + npmUnknownManager: (globalRoot, target) => + `This is a global install under ${globalRoot}, but the package manager that owns it could not be identified. Upgrade it yourself with that manager, e.g. \`npm install -g @prismshadow/penguin-cli@${target}\`.`, + windowsUnsupported: () => + "The official installer is a POSIX shell script and does not run on Windows. Re-install from the GitHub Releases page, or use a global npm install instead.", + windowsGlobalInstall: (command) => + [ + "On Windows, penguin cannot run your package manager for you: Node will not execute an npm/pnpm/yarn `.cmd` shim without a shell.", + "Run this yourself in a terminal, then reopen it:", + ` ${command}`, + ].join("\n"), + networkFailed: (url) => `Could not reach ${url}. Check your network and retry.`, + rateLimited: () => + "GitHub rate-limited the release lookup. Wait a few minutes and retry, or pass --release to skip the lookup.", + apiFailed: (status) => `The GitHub release lookup failed with HTTP ${status}.`, + apiMalformed: () => + "The GitHub release lookup returned an unexpected response with no usable version tag.", + installerFetchFailed: (url) => `Could not download the installer from ${url}.`, + }, header, chatHints: () => @@ -326,6 +421,57 @@ const zh: Messages = { host: "监听地址(其次取环境变量 HOST,缺省 127.0.0.1)", noOpen: "不自动打开浏览器", }, + update: { + desc: "原地升级当前的 PenguinHarness 安装", + check: "只报告当前版本与最新版本,不做任何修改", + releaseOpt: + "指定目标版本而不是最新版(如 v0.1.2 或 0.1.2);之所以叫 --release,是因为 -v/--version 是 CLI 自身的版本参数", + yes: "跳过确认提示", + checkReport: (current, latest) => `已安装 ${current} · 最新 ${latest}`, + upgradeAvailable: (target) => `有可用升级:执行 \`penguin update\` 安装 ${target}。`, + upToDate: (current) => `已是最新版本(${current}),无需升级。`, + targetIsOlder: (target) => `${target} 低于当前已安装的版本——这将是一次降级。`, + planTarball: (current, target, installDir, universal) => + [ + `升级 ${current} -> ${target}`, + ` 方式: 重新执行官方安装脚本(当前安装来自 tarball)`, + ` 安装目录:${installDir}${universal ? "(universal 包,不含内置 Node 运行时)" : ""}`, + ` 将替换: bin、lib、web${universal ? "" : "、node"}——数据目录不会被改动`, + ].join("\n"), + planNpm: (current, target, manager, command) => + [ + `升级 ${current} -> ${target}`, + ` 方式: ${manager} 全局安装(当前安装来自 ${manager})`, + ` 命令: ${command}`, + ` 数据目录不会被改动`, + ].join("\n"), + confirm: () => "确认继续?[y/N] ", + needsYes: () => "当前不在终端中,无法回答确认提示。请加 --yes 以非交互方式升级。", + cancelled: () => "已取消,未做任何修改。", + done: (version) => + `PenguinHarness ${version} 安装完成。在新 shell 中执行 \`penguin --version\` 确认。`, + failed: () => "升级失败;在可能的情况下已保留原有安装。", + sourceCheckout: () => + "当前 penguin 运行自源码检出,无需下载——请用 `git pull` 更新并重新构建(`pnpm install && pnpm -r build`)。", + unknownInstall: (modulePath) => + `无法判断当前 penguin 的安装方式(运行自 ${modulePath}),因此不会替换它。请用官方安装脚本重新安装,或用你当初使用的包管理器升级。`, + npmUnknownManager: (globalRoot, target) => + `这是位于 ${globalRoot} 的全局安装,但无法确定是哪个包管理器安装的。请自行用该包管理器升级,例如 \`npm install -g @prismshadow/penguin-cli@${target}\`。`, + windowsUnsupported: () => + "官方安装脚本是 POSIX shell 脚本,无法在 Windows 上运行。请从 GitHub Releases 页面重新安装,或改用 npm 全局安装。", + windowsGlobalInstall: (command) => + [ + "在 Windows 上,penguin 无法代你调用包管理器:Node 不会在没有 shell 的情况下执行 npm/pnpm/yarn 的 `.cmd` 包装脚本。", + "请在终端里自行执行下面的命令,然后重新打开终端:", + ` ${command}`, + ].join("\n"), + networkFailed: (url) => `无法访问 ${url}。请检查网络后重试。`, + rateLimited: () => + "GitHub 对版本查询做了限流。请等待几分钟后重试,或用 --release 跳过查询。", + apiFailed: (status) => `GitHub 版本查询失败,HTTP ${status}。`, + apiMalformed: () => "GitHub 版本查询返回了非预期的响应,其中没有可用的版本号。", + installerFetchFailed: (url) => `无法从 ${url} 下载安装脚本。`, + }, header, chatHints: () => diff --git a/packages/cli/src/index.ts b/packages/cli/src/index.ts index de396e0..9f47bd2 100644 --- a/packages/cli/src/index.ts +++ b/packages/cli/src/index.ts @@ -9,6 +9,7 @@ * penguin chat ... * penguin run --message ... * penguin server|web ... + * penguin update ... * Docs: packages/docs/content/cli.{zh,en}.md (site path /docs/cli). */ import "dotenv/config"; @@ -18,6 +19,7 @@ import { registerConfigCommand } from "./commands/config.js"; import { registerRunCommand } from "./commands/run.js"; import { registerChatCommand } from "./commands/chat.js"; import { registerServeCommands } from "./commands/serve.js"; +import { registerUpdateCommand } from "./commands/update.js"; import { defaultMessages } from "./i18n.js"; // Language comes from the PENGUIN_LANG env var (default en); used consistently for @@ -34,6 +36,7 @@ registerConfigCommand(program, t); registerRunCommand(program, t); registerChatCommand(program, t); registerServeCommands(program, t); +registerUpdateCommand(program, t); // Show help only when no subcommand is given (empty input); do not error. program.action(() => { diff --git a/packages/cli/test/update.test.ts b/packages/cli/test/update.test.ts new file mode 100644 index 0000000..5194ec8 --- /dev/null +++ b/packages/cli/test/update.test.ts @@ -0,0 +1,410 @@ +/** + * `penguin update`'s pure pieces: install-kind detection over the three real layouts, package- + * manager identification for global installs, version normalisation/comparison, the installer + * argv/env built for each combination of install dir and bundled runtime, and the decision the + * command makes before it touches anything (planUpdate) plus its confirmation gate. + * + * No network and no filesystem mutation — every function under test takes its inputs as arguments. + */ +import { describe, expect, it } from "vitest"; +import { Command } from "commander"; +import { + buildInstallerInvocation, + compareVersions, + confirmationMode, + detectInstall, + detectPackageManager, + globalInstallCommand, + installerUrl, + normalizeVersion, + planUpdate, + registerUpdateCommand, +} from "../src/commands/update.js"; +import { getMessages } from "../src/i18n.js"; + +describe("detectInstall (how this CLI was installed, from its own real path)", () => { + it("tarball: /lib/dist/index.js, the layout install.sh unpacks", () => { + expect(detectInstall("/home/me/.penguin/lib/dist/index.js")).toEqual({ + kind: "tarball", + installDir: "/home/me/.penguin", + }); + }); + + it("tarball: a non-default PENGUIN_INSTALL_DIR is read off the path, not the environment", () => { + expect(detectInstall("/opt/tools/penguin/lib/dist/index.js")).toEqual({ + kind: "tarball", + installDir: "/opt/tools/penguin", + }); + }); + + it("npm global: npm's own prefix layout", () => { + expect( + detectInstall("/usr/local/lib/node_modules/@prismshadow/penguin-cli/dist/index.js"), + ).toEqual({ kind: "npm", globalRoot: "/usr/local/lib/node_modules" }); + }); + + it("npm global: pnpm's global store, through the .pnpm virtual dir", () => { + const p = + "/home/me/.local/share/pnpm/global/5/node_modules/.pnpm/@prismshadow+penguin-cli@0.1.1/node_modules/@prismshadow/penguin-cli/dist/index.js"; + const info = detectInstall(p); + expect(info.kind).toBe("npm"); + expect(info.globalRoot).toContain(".pnpm"); + }); + + it("source checkout: the built dist inside the monorepo", () => { + expect(detectInstall("/home/me/code/penguin-harness/packages/cli/dist/index.js")).toEqual({ + kind: "source", + }); + }); + + it("source checkout: tsx running src directly", () => { + expect(detectInstall("/home/me/code/penguin-harness/packages/cli/src/index.ts")).toEqual({ + kind: "source", + }); + }); + + it("a checkout wins over the tarball shape, so a repo under a lib/ dir is never mistaken for an install", () => { + expect(detectInstall("/srv/lib/penguin-harness/packages/cli/dist/index.js")).toEqual({ + kind: "source", + }); + }); + + it("anything else is unknown rather than guessed", () => { + expect(detectInstall("/random/place/index.js").kind).toBe("unknown"); + expect(detectInstall("/home/me/.penguin/bin/penguin").kind).toBe("unknown"); + }); +}); + +describe("detectPackageManager (which manager owns a global node_modules root)", () => { + it("pnpm: the global store or the .pnpm virtual dir", () => { + expect(detectPackageManager("/home/me/.local/share/pnpm/global/5/node_modules")).toBe("pnpm"); + expect( + detectPackageManager("/home/me/.local/share/pnpm/global/5/node_modules/.pnpm/x/node_modules"), + ).toBe("pnpm"); + }); + it("npm: the usual prefixes", () => { + expect(detectPackageManager("/usr/local/lib/node_modules")).toBe("npm"); + expect(detectPackageManager("/home/me/.npm-global/lib/node_modules")).toBe("npm"); + }); + it("yarn and bun", () => { + expect(detectPackageManager("/home/me/.config/yarn/global/node_modules")).toBe("yarn"); + expect(detectPackageManager("/home/me/.bun/install/global/node_modules")).toBe("bun"); + }); + it("returns null when unrecognizable, so the caller prints a command instead of guessing", () => { + expect(detectPackageManager("/weird/place")).toBeNull(); + expect(detectPackageManager("")).toBeNull(); + }); +}); + +describe("globalInstallCommand", () => { + it("uses each manager's own global-install spelling", () => { + expect(globalInstallCommand("pnpm", "0.1.2")).toEqual({ + command: "pnpm", + args: ["add", "-g", "@prismshadow/penguin-cli@0.1.2"], + }); + expect(globalInstallCommand("npm", "0.1.2")).toEqual({ + command: "npm", + args: ["install", "-g", "@prismshadow/penguin-cli@0.1.2"], + }); + expect(globalInstallCommand("yarn", "0.1.2")).toEqual({ + command: "yarn", + args: ["global", "add", "@prismshadow/penguin-cli@0.1.2"], + }); + expect(globalInstallCommand("bun", "0.1.2")).toEqual({ + command: "bun", + args: ["add", "-g", "@prismshadow/penguin-cli@0.1.2"], + }); + }); +}); + +describe("normalizeVersion / compareVersions", () => { + it("accepts both v-prefixed and bare tags", () => { + expect(normalizeVersion("v0.1.2")).toBe("0.1.2"); + expect(normalizeVersion("0.1.2")).toBe("0.1.2"); + expect(normalizeVersion(" V0.1.2 ")).toBe("0.1.2"); + }); + it("equal versions compare equal regardless of the v prefix", () => { + expect(compareVersions("v0.1.1", "0.1.1")).toBe(0); + }); + it("orders newer above older, including across component widths", () => { + expect(compareVersions("0.1.2", "0.1.1")).toBe(1); + expect(compareVersions("0.1.1", "0.1.2")).toBe(-1); + expect(compareVersions("0.2.0", "0.1.9")).toBe(1); + expect(compareVersions("1.0.0", "0.99.99")).toBe(1); + expect(compareVersions("0.1.10", "0.1.9")).toBe(1); + }); + it("treats a missing component as 0", () => { + expect(compareVersions("1", "1.0.0")).toBe(0); + expect(compareVersions("1.1", "1.0.9")).toBe(1); + }); + it("a malformed tag can never look like an available upgrade", () => { + expect(compareVersions("not-a-version", "0.1.1")).toBe(-1); + expect(compareVersions("", "0.1.1")).toBe(-1); + }); + it("reads each component as far as it is numeric, which is what the docblock promises", () => { + // parseInt semantics, pinned so the comment and the behaviour cannot drift apart again: + // leading digits win, and a component with none counts as 0. + expect(compareVersions("0.1.2abc", "0.1.2")).toBe(0); + expect(compareVersions("0.1.x", "0.1.0")).toBe(0); + expect(compareVersions("0.1.x", "0.1.1")).toBe(-1); + }); + it("suffixes are invisible, so a pre-release compares equal to its release", () => { + // Documented limitation rather than a bug: this project publishes plain vX.Y.Z tags only. + expect(compareVersions("0.1.2-rc1", "0.1.2")).toBe(0); + expect(compareVersions("0.1.2-rc1", "0.1.1")).toBe(1); + }); +}); + +describe("installerUrl", () => { + it("resolves the latest release when no version is pinned", () => { + expect(installerUrl()).toBe( + "https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh", + ); + }); + it("pins a tag, normalising the v prefix", () => { + const expected = + "https://github.com/Prism-Shadow/penguin-harness/releases/download/v0.1.2/install.sh"; + expect(installerUrl("0.1.2")).toBe(expected); + expect(installerUrl("v0.1.2")).toBe(expected); + }); +}); + +describe("buildInstallerInvocation (preserves the shape of the install being upgraded)", () => { + const base = { + scriptPath: "/tmp/penguin-install-1.sh", + defaultInstallDir: "/home/me/.penguin", + }; + + it("default dir + bundled runtime: no flags, no env", () => { + expect( + buildInstallerInvocation({ + ...base, + installDir: "/home/me/.penguin", + hasBundledNode: true, + }), + ).toEqual({ args: ["/tmp/penguin-install-1.sh"], env: {} }); + }); + + it("no bundled node/ means the universal package: pass --universal so the runtime is not silently added", () => { + expect( + buildInstallerInvocation({ + ...base, + installDir: "/home/me/.penguin", + hasBundledNode: false, + }), + ).toEqual({ args: ["/tmp/penguin-install-1.sh", "--universal"], env: {} }); + }); + + it("a non-default install dir is passed through, or the upgrade would relocate the install", () => { + expect( + buildInstallerInvocation({ + ...base, + installDir: "/opt/penguin", + hasBundledNode: true, + }), + ).toEqual({ + args: ["/tmp/penguin-install-1.sh"], + env: { PENGUIN_INSTALL_DIR: "/opt/penguin" }, + }); + }); + + it("a pinned version becomes PENGUIN_VERSION, always v-prefixed for the release tag", () => { + expect( + buildInstallerInvocation({ + ...base, + installDir: "/home/me/.penguin", + hasBundledNode: true, + version: "0.1.2", + }).env, + ).toEqual({ PENGUIN_VERSION: "v0.1.2" }); + expect( + buildInstallerInvocation({ + ...base, + installDir: "/home/me/.penguin", + hasBundledNode: true, + version: "v0.1.2", + }).env, + ).toEqual({ PENGUIN_VERSION: "v0.1.2" }); + }); + + it("all three at once: custom dir, universal package, pinned version", () => { + expect( + buildInstallerInvocation({ + ...base, + installDir: "/opt/penguin", + hasBundledNode: false, + version: "v0.2.0", + }), + ).toEqual({ + args: ["/tmp/penguin-install-1.sh", "--universal"], + env: { PENGUIN_INSTALL_DIR: "/opt/penguin", PENGUIN_VERSION: "v0.2.0" }, + }); + }); +}); + +describe("planUpdate (what the command decides before it touches anything)", () => { + const base = { + current: "0.1.1", + target: "0.1.2", + modulePath: "/home/me/.penguin/lib/dist/index.js", + platform: "linux", + defaultInstallDir: "/home/me/.penguin", + }; + const tarball = { kind: "tarball", installDir: "/home/me/.penguin" } as const; + const npmGlobal = { kind: "npm", globalRoot: "/usr/local/lib/node_modules" } as const; + + it("--check reports the comparison and never upgrades, even when one is available", () => { + expect(planUpdate({ ...base, check: true, install: tarball })).toEqual({ + action: "report", + current: "0.1.1", + target: "0.1.2", + comparison: 1, + }); + }); + + it("--check reports a downgrade and an up-to-date install without upgrading either", () => { + expect(planUpdate({ ...base, target: "0.1.0", check: true, install: tarball })).toMatchObject({ + action: "report", + comparison: -1, + }); + expect(planUpdate({ ...base, target: "0.1.1", check: true, install: tarball })).toMatchObject({ + action: "report", + comparison: 0, + }); + }); + + it("--check reports even from a source checkout, where an upgrade would be refused", () => { + // Reporting changes nothing, so the layout must not gate it. + expect(planUpdate({ ...base, check: true, install: { kind: "source" } }).action).toBe("report"); + }); + + it("an install already on the target exits without touching anything", () => { + expect(planUpdate({ ...base, target: "0.1.1", install: tarball })).toEqual({ + action: "up-to-date", + current: "0.1.1", + }); + // Decided before the layout matters: even an unknown install is simply up to date. + expect(planUpdate({ ...base, target: "v0.1.1", install: { kind: "unknown" } }).action).toBe( + "up-to-date", + ); + }); + + it("a source checkout is refused, because overwriting a working tree destroys work", () => { + expect(planUpdate({ ...base, install: { kind: "source" } })).toEqual({ + action: "refuse", + reason: "source", + }); + }); + + it("an unrecognised layout is refused and names the path it was running from", () => { + expect( + planUpdate({ ...base, modulePath: "/random/place/index.js", install: { kind: "unknown" } }), + ).toEqual({ + action: "refuse", + reason: "unknown-install", + modulePath: "/random/place/index.js", + }); + }); + + it("a global install upgrades through its own manager", () => { + expect(planUpdate({ ...base, install: npmGlobal })).toEqual({ + action: "npm", + manager: "npm", + command: "npm", + args: ["install", "-g", "@prismshadow/penguin-cli@0.1.2"], + }); + }); + + it("a global install with an unidentifiable manager is refused rather than guessed", () => { + expect(planUpdate({ ...base, install: { kind: "npm", globalRoot: "/weird/place" } })).toEqual({ + action: "refuse", + reason: "unknown-manager", + globalRoot: "/weird/place", + target: "0.1.2", + }); + }); + + it("a tarball install re-runs the installer at the dir it was detected in", () => { + expect(planUpdate({ ...base, install: tarball })).toEqual({ + action: "tarball", + installDir: "/home/me/.penguin", + }); + expect( + planUpdate({ ...base, install: { kind: "tarball", installDir: "/opt/penguin" } }), + ).toEqual({ action: "tarball", installDir: "/opt/penguin" }); + // No installDir on the info (shouldn't happen, but the fallback is the default install dir). + expect(planUpdate({ ...base, install: { kind: "tarball" } })).toEqual({ + action: "tarball", + installDir: "/home/me/.penguin", + }); + }); + + it("Windows is refused on the tarball path: the installer is a POSIX shell script", () => { + expect(planUpdate({ ...base, platform: "win32", install: tarball })).toEqual({ + action: "refuse", + reason: "windows-installer", + }); + }); + + it("Windows is refused on the npm path too, handing over the exact command to run", () => { + // spawn() cannot run a .cmd shim without a shell, so this must not reach the spawn and + // fail with a generic message. + expect(planUpdate({ ...base, platform: "win32", install: npmGlobal })).toEqual({ + action: "refuse", + reason: "windows-global", + command: "npm install -g @prismshadow/penguin-cli@0.1.2", + }); + }); + + it("every refusal has a message in both languages", () => { + const plans = [ + planUpdate({ ...base, install: { kind: "source" } }), + planUpdate({ ...base, install: { kind: "unknown" } }), + planUpdate({ ...base, install: { kind: "npm", globalRoot: "/weird/place" } }), + planUpdate({ ...base, platform: "win32", install: tarball }), + planUpdate({ ...base, platform: "win32", install: npmGlobal }), + ]; + for (const lang of ["en", "zh"] as const) { + const t = getMessages(lang); + expect(plans.map((p) => p.action)).toEqual(Array(plans.length).fill("refuse")); + expect(t.update.sourceCheckout()).toBeTruthy(); + expect(t.update.unknownInstall("/x")).toContain("/x"); + expect(t.update.npmUnknownManager("/weird/place", "0.1.2")).toContain("/weird/place"); + expect(t.update.windowsUnsupported()).toBeTruthy(); + // The Windows global-install message is only useful if it carries the command verbatim. + expect(t.update.windowsGlobalInstall("pnpm add -g pkg@1")).toContain("pnpm add -g pkg@1"); + } + }); +}); + +describe("confirmationMode (the gate in front of every upgrade)", () => { + it("--yes proceeds without a prompt, interactive or not", () => { + expect(confirmationMode(true, true)).toBe("proceed"); + expect(confirmationMode(true, false)).toBe("proceed"); + }); + it("a non-TTY without --yes demands --yes instead of hanging on a prompt nobody can answer", () => { + expect(confirmationMode(undefined, false)).toBe("needs-yes"); + expect(confirmationMode(false, false)).toBe("needs-yes"); + expect(getMessages("en").update.needsYes()).toContain("--yes"); + expect(getMessages("zh").update.needsYes()).toContain("--yes"); + }); + it("a terminal without --yes gets the interactive prompt", () => { + expect(confirmationMode(undefined, true)).toBe("prompt"); + }); +}); + +describe("command registration", () => { + it("registers `update` with its three options in both languages", () => { + for (const lang of ["en", "zh"] as const) { + const program = new Command(); + registerUpdateCommand(program, getMessages(lang)); + const cmd = program.commands.find((c) => c.name() === "update"); + expect(cmd).toBeDefined(); + const flags = cmd?.options.map((o) => o.flags) ?? []; + expect(flags).toContain("--check"); + expect(flags).toContain("--release "); + expect(flags).toContain("-y, --yes"); + expect(cmd?.description()).toBeTruthy(); + } + }); +}); diff --git a/packages/core/package.json b/packages/core/package.json index eede7cb..1a07787 100644 --- a/packages/core/package.json +++ b/packages/core/package.json @@ -1,6 +1,6 @@ { "name": "@prismshadow/penguin-core", - "version": "0.1.0", + "version": "0.1.1", "type": "module", "description": "PenguinHarness core SDK: context_engine, OmniMessage protocol, LLM/Environment interfaces.", "license": "Apache-2.0", diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 76211ad..b8cd862 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -51,4 +51,4 @@ export { Agent, createAgent } from "./agent.js"; export type { CreateAgentOptions, CreateSessionOptions, ResumeSessionOptions } from "./agent.js"; /** SDK version number. */ -export const VERSION = "0.1.0"; +export const VERSION = "0.1.1"; diff --git a/packages/docs/content/cli.en.md b/packages/docs/content/cli.en.md index 730c75c..7d98055 100644 --- a/packages/docs/content/cli.en.md +++ b/packages/docs/content/cli.en.md @@ -143,4 +143,29 @@ penguin web Port / host priority: command-line option > the `PORT` / `HOST` env vars (including `.env`) > defaults. +## penguin update + +Upgrades this install in place, using the mechanism it was installed with. The install kind is detected from the real path of the running CLI, never guessed. + +```bash +penguin update --check # report versions only +penguin update # upgrade to the latest release, after confirming +``` + +| Option | Description | +| --- | --- | +| `--check` | Only report the installed and latest versions; change nothing. Exit code is 0 either way | +| `--release ` | Target a specific release instead of the latest (`v0.1.2` or `0.1.2`); older tags are allowed and reported as a downgrade | +| `-y, --yes` | Skip the confirmation prompt | + +The target flag is `--release`, not `--version`, because `-v, --version` is the CLI's own version flag and would take precedence. + +| Install kind | How it upgrades | +| --- | --- | +| Tarball (`install.sh`, default `~/.penguin`) | Re-runs the official installer, preserving the install dir and whether the package bundles a Node runtime | +| Global npm/pnpm/yarn/bun install | Runs that manager's global install of `@prismshadow/penguin-cli@`; if the manager cannot be identified, prints the command instead of guessing | +| Source checkout | Refused — update it with `git pull` and a rebuild | + +Without `-y` the command prints exactly what it will do — mechanism, target version and install dir — and asks for confirmation; when stdin is not a terminal it requires `--yes` rather than waiting on a prompt nobody can answer. The latest version comes from the GitHub Releases API. **The data root is never touched**: an upgrade only replaces `bin`, `lib`, `web` and `node`. Neither path upgrades in place on Windows: the installer is a POSIX shell script, and a global install cannot be driven from here because Node will not execute an `npm`/`pnpm` `.cmd` shim without a shell — so the command prints the exact command to run yourself instead. + See also: [Configuration Reference](/configuration), [Models & Providers](/models). diff --git a/packages/docs/content/cli.zh.md b/packages/docs/content/cli.zh.md index 7fcc81c..73a3fca 100644 --- a/packages/docs/content/cli.zh.md +++ b/packages/docs/content/cli.zh.md @@ -143,4 +143,29 @@ penguin web 端口 / 地址优先级:命令行选项 > 环境变量 `PORT` / `HOST`(含 `.env`)> 默认值。 +## penguin update + +原地升级当前安装,并沿用它当初的安装方式。安装方式由运行中 CLI 的真实路径判定,不做猜测。 + +```bash +penguin update --check # 只报告版本 +penguin update # 确认后升级到最新版 +``` + +| 选项 | 说明 | +| --- | --- | +| `--check` | 只报告已安装版本与最新版本,不做任何修改;两种情况下退出码均为 0 | +| `--release ` | 指定目标版本而不是最新版(`v0.1.2` 或 `0.1.2`);允许低于当前版本,会明确提示为降级 | +| `-y, --yes` | 跳过确认提示 | + +目标版本参数叫 `--release` 而不是 `--version`,因为 `-v, --version` 是 CLI 自身的版本参数,会优先生效。 + +| 安装方式 | 升级方式 | +| --- | --- | +| tarball(`install.sh`,默认 `~/.penguin`) | 重新执行官方安装脚本,并保持原安装目录以及是否内置 Node 运行时 | +| npm/pnpm/yarn/bun 全局安装 | 用该包管理器全局安装 `@prismshadow/penguin-cli@<目标版本>`;无法确定包管理器时,只打印命令而不猜测 | +| 源码检出 | 拒绝执行——请用 `git pull` 更新并重新构建 | + +不带 `-y` 时,命令会先打印它将要做什么——方式、目标版本与安装目录——再请求确认;当 stdin 不是终端时,它要求显式加 `--yes`,而不是卡在无人能回答的提示上。最新版本取自 GitHub Releases API。**数据目录不会被改动**:升级只替换 `bin`、`lib`、`web` 与 `node`。两条路径在 Windows 上都不做原地升级:安装脚本是 POSIX shell 脚本,而全局安装也无法由此驱动——Node 不会在没有 shell 的情况下执行 `npm`/`pnpm` 的 `.cmd` 包装脚本——因此命令会直接打印出应当由你自己执行的命令。 + 相关文档:[配置参考](/configuration)、[模型与 Provider](/models)。 diff --git a/packages/docs/content/skills.en.md b/packages/docs/content/skills.en.md index ea65a0c..4dc21e0 100644 --- a/packages/docs/content/skills.en.md +++ b/packages/docs/content/skills.en.md @@ -60,6 +60,7 @@ The built-in Skills, by group (the group manifest is `SKILL_GROUPS` in `packages | --- | --- | --- | | Office Productivity | `data-analysis` | Complete data-analysis tasks with bounded evidence inspection, explicit answer-changing decisions, native artifact handling and final output verification | | | `firecrawl` | Web search and page scraping into clean markdown via the Firecrawl API | +| | `bento-slides` | Author and edit Bento presentations: single-file `.bento.html` decks whose document is JSON, mapping material to charts, morph transitions and state slides | | Software Development | `web-design` | Penguin visual language for generated web pages and app UIs: design tokens, components, light/dark themes and chat layouts | | | `software-engineering` | Complete software-engineering tasks: investigate and review code, implement fixes, features and refactors with minimal scope, validate changes, and report verified outcomes | | AI App Development | `penguin-sdk` | Build AI and RAG apps on the SDK: the createSession/run streaming loop plus a complete retrieval recipe with chunk-revealing citations | diff --git a/packages/docs/content/skills.zh.md b/packages/docs/content/skills.zh.md index af430d6..8c994bb 100644 --- a/packages/docs/content/skills.zh.md +++ b/packages/docs/content/skills.zh.md @@ -60,6 +60,7 @@ Skill 库以 npm 包 `@prismshadow/penguin-skills` 发布,tarball 直接携带 | --- | --- | --- | | 办公效率 | `data-analysis` | 以有界的证据检查、显式的改答案决策、原生产物处理与最终输出校验完成数据分析任务 | | | `firecrawl` | 经 Firecrawl API 做网络搜索与页面抓取,产出干净的 Markdown | +| | `bento-slides` | 制作与编辑 Bento 演示文稿:单文件 `.bento.html`、文档即 JSON,把素材映射到图表、morph 转场与状态页 | | 软件开发 | `web-design` | 生成网页与应用界面的 Penguin 视觉语言:设计令牌、组件配方、明暗主题与聊天布局 | | | `software-engineering` | 完成软件工程任务:调查与审查代码,以最小改动实现修复、特性与重构,验证改动并报告经过确认的结果 | | AI 应用开发 | `penguin-sdk` | 基于 SDK 构建 AI 与 RAG 应用:createSession/run 流式循环,外加带可溯源引用的完整检索配方 | diff --git a/packages/docs/package.json b/packages/docs/package.json index 14d7eed..e364c1a 100644 --- a/packages/docs/package.json +++ b/packages/docs/package.json @@ -1,6 +1,6 @@ { "name": "@prismshadow/penguin-docs", - "version": "0.1.0", + "version": "0.1.1", "private": true, "type": "module", "description": "PenguinHarness documentation site (React + Vite + Tailwind CSS): bilingual zh/en Markdown pages with light/dark themes and per-page Copy Markdown, deployed to GitHub Pages under /docs/ next to the landing page.", diff --git a/packages/docs/src/styles.css b/packages/docs/src/styles.css index 050ba22..6e47484 100644 --- a/packages/docs/src/styles.css +++ b/packages/docs/src/styles.css @@ -160,9 +160,26 @@ } .md-body a { @apply text-brand-700 underline decoration-brand-300 underline-offset-2 transition-colors hover:text-brand-600 dark:text-brand-300 dark:decoration-brand-700; + /* `anywhere` rather than `word-break: break-all`: it only breaks a token that would otherwise + overflow, so long URLs wrap cleanly (filling each line) while short Latin words in mixed + CJK/Latin link text never split mid-word; unlike `break-word` it also counts the break + opportunities toward min-content sizing, so a long link can't blow out flex/table layouts. */ + overflow-wrap: anywhere; } +/* Font size only, so fenced blocks keep the size they have always had; the visible chrome below + is scoped to inline code. */ .md-body code { - @apply rounded bg-gray-100 px-1 py-0.5 text-[0.85em] text-gray-800 dark:bg-gray-800 dark:text-gray-200; + @apply text-[0.85em]; +} +/* Inline code: a hairline outline over a barely-there brand wash, replacing the solid gray fill + this used to carry — in prose dense with `code` that fill read as a row of heavy blocks. The + chrome is scoped to inline code because the `.md-body pre code` reset below strips background, + color and padding inside fenced blocks but would not strip a border. Prose often holds long + unbroken paths and identifiers, so inline code gets the same `anywhere` treatment as links. + Kept identical to the landing site: penguin.ooo and penguin.ooo/docs share one design language. */ +.md-body :not(pre) > code { + @apply rounded-md border border-brand-200/60 bg-brand-50/40 px-[0.3em] py-[0.1em] text-[0.875em] text-gray-900 dark:border-brand-900/70 dark:bg-brand-950/50 dark:text-gray-100; + overflow-wrap: anywhere; } .md-body pre { @apply overflow-x-auto rounded-lg border border-gray-200 bg-gray-50 p-3 text-[13px] leading-6 dark:border-gray-800 dark:bg-gray-900; diff --git a/packages/landing/content/blog/gemini-3-6-in-penguinharness.en.md b/packages/landing/content/blog/gemini-3-6-in-penguinharness.en.md new file mode 100644 index 0000000..56dafbb --- /dev/null +++ b/packages/landing/content/blog/gemini-3-6-in-penguinharness.en.md @@ -0,0 +1,120 @@ +--- +title: "PenguinHarness 0.1.1: Gemini 3.6 Flash support, for building agents faster and better" +date: 2026-07-22 +category: news +excerpt: Google reports Gemini 3.6 Flash at 63.9% on MLE-Bench, against 49.7% for 3.5 Flash. MLE-Bench scores machine-learning engineering — an agent doing the work of building and tuning ML systems, which is exactly what PenguinHarness exists to do. The model is in the catalog today, at a lower price than the Flash it replaces. Here is the case, and the rest of the 0.1.1 release. +--- + +Google announced [Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber](https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/) on July 21, 2026. One number in that announcement matters more to PenguinHarness than everything else in it: **MLE-Bench, 63.9% against 3.5 Flash's 49.7%**. + +## Why that one number + +MLE-Bench scores machine-learning *engineering* — an agent doing the work of building, training and tuning an ML system, not answering questions about one. Google's own phrasing for the result is a "significant improvement in ML Research, as seen in MLE Bench (63.9% vs. 49.7%)". That is a 14.2-point gain, the largest move of the three percentage-scored panels in Google's chart below, and it lands well above the 42.6% the same chart records for the previous Pro-generation model, 3.1 Pro. + +PenguinHarness exists so that **agents build agents — faster, better, cheaper**. Every loop the product runs is machine-learning engineering in miniature: stand a model up, hand an agent a task, score it against a private rubric, read the failures, tune, redeploy, measure again. A model that is markedly better at precisely that work, priced in the Flash tier, is the strongest model yet for what this harness does all day. That is the argument; the rest of this post is the evidence, and then the rest of the release. + +One caveat up front, and it holds for every figure below: these are **Google's numbers, from Google's evaluation methodology**. We have not independently reproduced any of them. + +## The rest of the scoreboard + +![Gemini 3.6 Flash evaluation chart: DeepSWE v1.1, MLE-Bench, GDPval-AA v2 and OSWorld-Verified, each comparing Gemini 3.1 Pro, 3.5 Flash and 3.6 Flash](/blog-assets/gemini-3-6-flash-evals.webp) + +*Figure by Google, reproduced from its announcement post [Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber](https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/) (July 21, 2026). All numbers and the evaluation methodology are Google's, not ours.* + +MLE-Bench does not stand alone. Against 3.5 Flash, Google reports: + +| Benchmark | What it measures | 3.5 Flash | 3.6 Flash | +| ---------------- | --------------------------------- | --------: | --------: | +| MLE-Bench | Machine-learning engineering | 49.7% | 63.9% | +| DeepSWE v1.1 | Long-horizon software engineering | 37% | 49% | +| GDPval-AA v2 | Knowledge work | 1349 | 1421 | +| OSWorld-Verified | Computer use | 78.4% | 83.0% | + +The three supporting rows are the ones an ML-engineering loop actually leans on. Long-horizon software engineering is how the agent edits the training config, the dataset builder and the serving flags without thrashing — Google attributes the DeepSWE jump to "higher precision with fewer unwanted code edits and reduced execution loops", the failure mode anyone who has watched an agent churn through a repository will recognize. Computer use is now a built-in client-side tool in the Gemini API and Gemini Enterprise. And the model ships with enhanced Frontier Safety safeguards for CBRN and cyber-offense misuse that Google says make it "substantially more resistant to jailbreaks" while minimizing refusals for beneficial uses. + +## The cheaper leg: fewer tokens, fewer steps, lower price + +Google's framing of the release lands squarely on what a harness like this one does all day: "Developers and customers building production AI agents need higher token efficiency, lower latency, and more reliable performance." An agent loop is not one long completion — it is dozens of short round trips, each carrying a tool schema, a growing transcript and a reasoning budget. Efficiency per step is the whole cost model. + +Google reports 3.6 Flash consuming **17% fewer output tokens than 3.5 Flash** on the Artificial Analysis Index, with "up to 65%" observed on some benchmarks such as DeepSWE by Datacurve, and says it "takes fewer reasoning steps and tool calls to accomplish multi-step workflows." That efficiency "is also combined with a lower price than 3.5 Flash": **$1.50 per million input tokens and $7.50 per million output tokens**. Google's own conclusion is the one that matters here — the combination "reduces the overall cost per agentic task, making agents more cost-effective to build and run." + +Fewer tokens, fewer steps, lower unit price. For a self-improving loop that re-runs a whole benchmark suite on every round, the three compound. + +## What ships in PenguinHarness today + +One day after the announcement, the catalog carries both generally available models, on two routes each: + +| Provider group | Model id | Context | Vision | +| -------------- | ------------------------------ | --------: | ------ | +| Google Gemini | `gemini-3.6-flash` | 1,048,576 | yes | +| Google Gemini | `gemini-3.5-flash-lite` | 1,048,576 | yes | +| OpenRouter | `google/gemini-3.6-flash` | 1,048,576 | yes | +| OpenRouter | `google/gemini-3.5-flash-lite` | 1,048,576 | yes | + +The Google-endpoint rows are auto-routed by model id, so a `GEMINI_API_KEY` is all they need. The OpenRouter rows carry the OpenAI client type and the gateway's base URL inline, so they need nothing but an OpenRouter key. Either way, the fastest path is the **Models** page in the Web App: find the row, add the key, done. From the CLI it is one command: + +```bash +penguin config model add --provider google --model-id gemini-3.6-flash --api-key +penguin config model list +``` + +Two related corrections landed with them. Gemini pricing is now recorded with the vendor's real cache-hit rate rather than the input price repeated into the cache bucket — $0.15 per million cached input tokens for 3.6 Flash, $0.03 for 3.5 Flash-Lite — so the Cost center no longer overstates cache-heavy spend by an order of magnitude. And `google/gemini-3.5-flash` had its context window recorded as 1,000,000; the real figure, on both the gateway and the direct endpoint, is 1,048,576. + +### 3.5 Flash-Lite, the fan-out half + +The second model in the catalog is aimed at volume rather than depth. Gemini 3.5 Flash-Lite is the fastest model in the 3.5 series as measured by Artificial Analysis, running at **350 output tokens per second**, priced at **$0.30 per million input tokens and $2.50 per million output tokens**. Google positions it for high-throughput work like agentic search and document processing, with configurable thinking levels so the same model can be driven cheap-and-shallow for bulk tasks or pushed higher for multi-step subagent workloads. Computer use is a built-in tool here too. + +Against the previous Flash-Lite generation, Google reports Terminal-Bench 2.1 at **54% vs 31%**, long context on GDM-MRCR v2 at **72.2% vs 60.1%**, and GDPval-AA v2 at **1140 vs 642**. On several agentic and coding evals it even passes 3 Flash — SWE-Bench Pro **54.2% vs 49.6%**, OSWorld-Verified **74.0% vs 65.1%**. + +That is a very good fit for the subagent pattern PenguinHarness runs on: a capable parent model planning, a cheap fast model fanning out. Google's own post shows the same shape — 3.5 Flash-Lite generating design concepts "alongside 3.6 Flash as the master agent." + +The third model in Google's announcement, 3.5 Flash Cyber, is deliberately out of reach: Google says it will be available exclusively to governments and trusted partners through CodeMender, as part of a limited-access pilot program. It is not something PenguinHarness — or anyone else — will be pointing a base URL at. + +## The rest of 0.1.1 + +The Gemini rows are one line of a much larger catalog refresh, and the catalog is one surface of a release that touched most of them. + +### Models and core + +The SDK moved to **AgentHub 0.4.1**, a type-compatible upgrade whose new supported-model registry — model, base URL and client triples with modalities, context windows and per-million pricing — became the authoritative source for a catalog diff. The catalog grew from 59 to 70 entries: the two Gemini rows above on both routes, Claude Fable 5 and Claude Sonnet 5 on Anthropic, Kimi K3 on Moonshot, Kimi K2.6, Qwen3.6 35B A3B and GLM 5.1 on OpenRouter, and the same three on SiliconFlow. The last three ship unpriced on purpose: no source publishes their rates, and a guessed number is worse than none. Every context window, vision flag and price came from the registry rather than a vendor marketing page. + +Three fixes matter if you run agents against local or strict endpoints: + +- **Empty tool lists no longer go on the wire.** Strict OpenAI-compatible servers reject `tools: []` outright — vLLM answers `400 … tools must not be an empty array`. Every tool-less request the harness makes (the connectivity probe, session-title generation, the vision describer) used to hit that. The field is now omitted entirely when the list is empty. +- **Max output tokens is a per-model setting.** A 32k-context model served locally would refuse requests because the agent-level default asked for 32,000 output tokens. The Models page and `penguin config model add --max-tokens` now take a per-model cap that applies ahead of the agent default, and out-of-band requests take the smaller of the two. +- **The default system prompt gained guardrails.** Agents that freed a busy port by killing its listener sometimes killed the harness's own services; the prompt now says never kill a process you did not start, and pick another free port instead. On a 401/403 or invalid-key error the agent retries once, then stops and asks you to update the key outside the conversation — secret values do not belong in a chat transcript. Existing agents keep their current prompt; new ones get the rules. + +Two runtime settings changed shape. The thinking level moved out of the Models page and into a compact picker next to the model selector in the chat draft (`low` / `medium` / `high` / `xhigh`), writing through to Agent settings so the session created on send uses it. And subagents now inherit the parent session's resolved `(provider, model_id)` pair and effective thinking level instead of falling back to the project default — an explicit pair in the tool call still wins. + +### Web App + +The chat sidebar now groups conversations **by Workspace** by default, labeled by directory basename, ordered by newest session, with auto-created temp workspaces collapsed into a single group instead of one per session. Groups can be pinned to the top, collapse state persists per Project, sessions created by subagents and scheduled tasks file into their own folders, and each group pages in more sessions on demand rather than fetching an unbounded list. Agent grouping is still one toggle away. + +The model dropdown now lists models that actually have a configured key first, with the rest one click below. The collapsed sidebar became a full eight-entry navigation rail with bilingual tooltips (Benchmark was previously missing entirely). Custom provider groups and Agents render initial-letter avatars on a tinted background derived from the name, holding WCAG AA contrast in both themes, so same-named models across groups stop being indistinguishable. + +Chat rendering got a pass: links open in a new tab, long URLs and inline code wrap at the container edge without splitting Latin words inside CJK prose, wide tables scroll inside the message instead of widening the page, expanded subagent conversations render below the tool call's own output, and three mobile dropdowns that used to overflow the viewport by up to 143px now stay inside it. The Cost center's daily-token tooltip follows the pointer and shows the cache hit rate; the copy-message task-stats line, previously hardcoded Chinese, now goes through the dictionaries like everything else. + +### Skills + +Three new skills join the AI App Development group — **vllm**, **ollama** and **llamafactory** — so an agent can stand up and tune the models it builds on, not just call them. Both serving skills share a guided workflow: ask which model to serve, ask which engine the user prefers, serve, verify, then register the endpoint with the CLI. The `penguin-cli` skill (now v5) and `penguin-sdk` carry the hard rule they lean on: configuring Penguin's own model uses the default data root, but models configured for an app under development must go into the app's own project directory. There is a whole [practice post](/blog/natural-language-training-loop) on what changes once an agent holds all three — you stop typing the commands and start describing the outcome. + +A fourth skill, **bento-slides**, joins the Office Productivity group: ask for a presentation and the agent authors a real Bento deck — one self-contained `.bento.html` whose document is JSON — mapping your material onto charts, morph transitions and state slides instead of a wall of bullets. It is adapted, with attribution, from the Bento project's own MIT-licensed skill. + +`agenthub-models` tracks the 0.4.1 upgrade: the new supported-model registry, the config parameters a client may now reject outright, and the Gemini 3.6 / Kimi K3 / GLM-5.2 families with their reasoning-effort knobs. + +### Sites, docs and tooling + +The docs and landing navbars are now literally the same layout — same container width, same logo block, same right cluster — after drifting apart into two near-identical implementations. The blog gained the **Tech practice** category, pinned posts, and per-post metadata (locale-formatted date, author line, copy-link button). Both READMEs and the landing page now list the built-in Skills where people actually look for them. + +The README roadmap gained two items — Agent company and templates, and company-level self evolving — and the self-improvement example under `examples/` was reworked into two runnable scripts that let a local open-weight model score itself, edit its own files and re-run, instead of a fixed illustrative transcript. On the hardening side, the server now validates paging and date query parameters instead of trusting them, and two previously uncovered core modules picked up unit tests. + +Upgrading is now one command. `penguin update` resolves the newest release, tells you exactly what it is about to do, and upgrades in place using the mechanism your install actually came from — re-running the official installer for a tarball install (keeping your install dir and your choice of bundled runtime), or the right global install for an npm/pnpm/yarn/bun one. It refuses to touch a source checkout, and it never touches your data dir. `penguin update --check` reports the versions and changes nothing. + +## Get it + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +penguin web +``` + +Then open the Models page, drop in a Gemini or OpenRouter key, and pick `gemini-3.6-flash`. The full release notes live in [`changelog/0.1.1/`](https://github.com/Prism-Shadow/penguin-harness/blob/main/changelog/0.1.1/README.md). diff --git a/packages/landing/content/blog/gemini-3-6-in-penguinharness.zh.md b/packages/landing/content/blog/gemini-3-6-in-penguinharness.zh.md new file mode 100644 index 0000000..7aa87d4 --- /dev/null +++ b/packages/landing/content/blog/gemini-3-6-in-penguinharness.zh.md @@ -0,0 +1,120 @@ +--- +title: PenguinHarness 0.1.1 更新:支持 Gemini 3.6 Flash,更快更好地构建 Agent +date: 2026-07-22 +category: news +excerpt: Google 给出的 Gemini 3.6 Flash MLE-Bench 成绩是 63.9%,3.5 Flash 是 49.7%。MLE-Bench 衡量的是机器学习工程能力——Agent 亲手构建、训练、调优一套 ML 系统,而这正是 PenguinHarness 要做的事。该模型今天已在模型目录中,价格还低于它取代的那一代 Flash。本文讲清这个判断,以及 0.1.1 版本的其余更新。 +--- + +Google 在 2026 年 7 月 21 日发布了 [Gemini 3.6 Flash、3.5 Flash-Lite 与 3.5 Flash Cyber](https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/)。这次发布里有一个数字,对 PenguinHarness 的意义超过其余全部:**MLE-Bench 63.9%,而 3.5 Flash 是 49.7%**。 + +## 为什么是这个数字 + +MLE-Bench 衡量的是机器学习*工程*能力——Agent 亲手构建、训练、调优一套 ML 系统,而不是回答关于它的问题。Google 自己的说法是,这体现了「ML Research 上的显著提升,见 MLE Bench(63.9% vs. 49.7%)」。这是 14.2 个百分点的提升,是下面那张图中三项百分制基准里幅度最大的一项,并且明显高于同一张图给出的上一代 Pro 级模型 3.1 Pro 的 42.6%。 + +PenguinHarness 存在的意义,就是**让 Agent 构建 Agent——更快、更好、更便宜**。这个产品跑的每一个循环,本质上都是一次小型的机器学习工程:把模型部署起来,交给 Agent 一个任务,用私有 Rubric 打分,读失败原因,调优,重新部署,再测一次。一个在这件事上明显更强、又落在 Flash 价格档的模型,就是迄今最适合这套 Harness 的模型。这是本文的论点;后面是证据,再后面是这个版本的其余更新。 + +有一点先说清楚,且对下文每一个数字都成立:这些都是 **Google 自己的数据、按 Google 自己的评测方法得出**,我们没有独立复现其中任何一项。 + +## 其余的成绩单 + +![Gemini 3.6 Flash 评测图:DeepSWE v1.1、MLE-Bench、GDPval-AA v2 与 OSWorld-Verified 四项,逐项对比 Gemini 3.1 Pro、3.5 Flash 与 3.6 Flash](/blog-assets/gemini-3-6-flash-evals.webp) + +*图片版权归 Google 所有,转载自其发布博客 [Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber](https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/)(2026 年 7 月 21 日)。图中全部数据与评测方法均出自 Google,而非我们自测。* + +MLE-Bench 不是孤证。对比 3.5 Flash,Google 给出的数据是: + +| 基准 | 衡量的能力 | 3.5 Flash | 3.6 Flash | +| ---------------- | ------------ | --------: | --------: | +| MLE-Bench | 机器学习工程 | 49.7% | 63.9% | +| DeepSWE v1.1 | 长程软件工程 | 37% | 49% | +| GDPval-AA v2 | 知识工作 | 1349 | 1421 | +| OSWorld-Verified | 计算机操作 | 78.4% | 83.0% | + +另外三行恰好是一个机器学习工程循环真正依赖的能力。长程软件工程决定了 Agent 能不能不反复折腾地改好训练配置、数据集脚本与部署参数——Google 把 DeepSWE 的提升归因于「更高的精确度、更少的多余代码改动和更少的执行循环」,任何看过 Agent 在代码仓库里来回折腾的人都认得这个失败模式。计算机操作现在是 Gemini API 与 Gemini Enterprise 中的内置客户端工具。模型还带上了针对 CBRN 与网络攻击滥用的 Frontier Safety 加固,Google 称其「明显更难被越狱」,同时减少了对正当用途的拒答。 + +## 「更便宜」这一条:更少 Token、更少步数、更低单价 + +Google 对这次发布的定位,恰好就是 Harness 每天在做的事:「构建生产级 AI Agent 的开发者与客户,需要更高的 Token 效率、更低的延迟和更可靠的表现。」Agent 循环不是一次长补全,而是几十次短往返,每一次都带着工具 Schema、不断增长的对话记录和一份推理预算。单步效率就是全部的成本模型。 + +按 Google 的说法,在 Artificial Analysis Index 上 3.6 Flash 比 3.5 Flash **少消耗 17% 的输出 Token**,在 Datacurve 的 DeepSWE 等部分基准上观察到「最高 65%」,并且「完成多步工作流所需的推理步数与工具调用更少」。这种效率提升「同时还配上了低于 3.5 Flash 的价格」:**输入 $1.50 / 百万 Token,输出 $7.50 / 百万 Token**。Google 自己的结论正是这里最要紧的一句——这套组合「降低了每个 Agent 任务的总体成本,让 Agent 的构建与运行更划算」。 + +更少的 Token、更少的步数、更低的单价。对于每一轮都要重跑一整套评测集的自我进化循环来说,这三者是叠乘关系。 + +## PenguinHarness 里现在能用到什么 + +发布一天之后,目录里已经收录了两个公开可用的模型,各有两条接入路径: + +| 提供方分组 | 模型 ID | 上下文 | 视觉 | +| ------------- | ------------------------------ | --------: | ---- | +| Google Gemini | `gemini-3.6-flash` | 1,048,576 | 支持 | +| Google Gemini | `gemini-3.5-flash-lite` | 1,048,576 | 支持 | +| OpenRouter | `google/gemini-3.6-flash` | 1,048,576 | 支持 | +| OpenRouter | `google/gemini-3.5-flash-lite` | 1,048,576 | 支持 | + +Google 直连的两行由模型 ID 自动路由,只需要一个 `GEMINI_API_KEY`;OpenRouter 的两行已经内联了 OpenAI 客户端类型与网关 base URL,只需要一个 OpenRouter Key。最快的路径是 Web App 的 **Models** 页面:找到那一行,填 Key,结束。命令行则是一条命令: + +```bash +penguin config model add --provider google --model-id gemini-3.6-flash --api-key +penguin config model list +``` + +随之落地的还有两处修正。Gemini 的价格现在按厂商真实的缓存命中价记录,而不是把输入价重复填进缓存桶——3.6 Flash 每百万缓存输入 Token $0.15,3.5 Flash-Lite $0.03——费用中心因此不再把缓存密集型开销高估一个数量级。另外 `google/gemini-3.5-flash` 的上下文窗口此前记为 1,000,000,网关与直连端点的真实值都是 1,048,576,现已更正。 + +### 3.5 Flash-Lite:负责扇出的那一半 + +目录里的第二个模型,目标是吞吐而不是深度。按 Artificial Analysis 的测量,Gemini 3.5 Flash-Lite 是 3.5 系列中最快的模型,达到 **350 输出 Token/秒**,定价 **输入 $0.30 / 百万 Token、输出 $2.50 / 百万 Token**。Google 把它定位在 Agent 检索、文档处理这类高吞吐场景,并支持配置思考等级——同一个模型既可以压到低延迟低成本跑批量任务,也可以调高思考等级承担多步子 Agent 负载。计算机操作在这里同样是内置工具。 + +与上一代 Flash-Lite 相比,Google 给出的数据是:Terminal-Bench 2.1 **54% 对 31%**,长上下文 GDM-MRCR v2 **72.2% 对 60.1%**,GDPval-AA v2 **1140 对 642**。在若干 Agent 与编码评测上它甚至超过 3 Flash——SWE-Bench Pro **54.2% 对 49.6%**,OSWorld-Verified **74.0% 对 65.1%**。 + +这非常契合 PenguinHarness 依赖的子 Agent 模式:能力更强的父模型负责规划,便宜且快的模型负责扇出执行。Google 自己的博客展示的也是同一形态——3.5 Flash-Lite「与作为主控 Agent 的 3.6 Flash 协同」批量生成设计方案。 + +发布中的第三个模型 3.5 Flash Cyber 则刻意不对外开放:Google 表示它将仅通过 CodeMender、以限量试点计划的形式提供给政府与可信合作伙伴。PenguinHarness——以及任何其他工具——都不会有机会把 base URL 指过去。 + +## 0.1.1 的其余更新 + +Gemini 这两行只是一次更大规模目录刷新中的一条,而目录也只是这个版本触及的诸多面之一。 + +### 模型与内核 + +SDK 升级到 **AgentHub 0.4.1**。这是一次类型兼容的升级,它新增的受支持模型注册表——模型、base URL 与 client 三元组,附带模态、上下文窗口和每百万 Token 价格——成了这次目录比对的权威来源。目录条目从 59 条增加到 70 条:上文两个 Gemini 模型的两条路径,Anthropic 的 Claude Fable 5 与 Claude Sonnet 5,Moonshot 的 Kimi K3,OpenRouter 的 Kimi K2.6、Qwen3.6 35B A3B 与 GLM 5.1,以及 SiliconFlow 上的同样三个。最后三个刻意不带价格:没有任何来源公布它们的费率,而猜一个数字比留空更糟。所有上下文窗口、视觉标记与价格都来自注册表,而不是厂商的宣传页。 + +如果你把 Agent 跑在本地或严格的端点上,有三处修复值得注意: + +- **空工具列表不再发到线上。** 严格的 OpenAI 兼容服务会直接拒绝 `tools: []`——vLLM 会回 `400 … tools must not be an empty array`。Harness 所有不带工具的请求(连通性探测、会话标题生成、视觉描述)此前都会撞上它。现在列表为空时该字段被整个省略。 +- **最大输出 Token 成为模型级设置。** 本地部署的 32k 上下文模型会直接拒绝请求,因为 Agent 级默认值要求 32,000 个输出 Token。Models 页面与 `penguin config model add --max-tokens` 现在支持按模型设置上限,其优先级高于 Agent 默认值,带外请求则取两者中较小的一个。 +- **默认系统提示词加了护栏。** Agent 为了腾出被占端口而杀掉监听进程时,有时杀掉的是 Harness 自己的服务;提示词现在要求:绝不杀死不是自己启动的进程,端口被占就另选一个空闲端口。遇到 401/403 或无效 Key 时,Agent 最多重试一次,随后停止并请用户在对话之外更新 Key——密钥不该出现在聊天记录里。已有 Agent 保留当前提示词,新建的 Agent 才带上这些规则。 + +还有两项运行时设置换了形态。思考等级从 Models 页面挪到了对话草稿区、模型选择器旁的紧凑选择器(`low` / `medium` / `high` / `xhigh`),并即时写回 Agent 设置,使发送时创建的会话就用新值。子 Agent 现在继承父会话已解析的 `(provider, model_id)` 二元组与生效的思考等级,不再回落到 Project 默认模型;工具调用里显式给出的完整二元组仍然优先。 + +### Web App + +聊天侧边栏默认**按 Workspace 分组**:以目录名作为标签、按最新会话排序,自动创建的临时 Workspace 收敛成单个分组而不是一个会话一个分组。分组可以置顶,折叠状态按 Project 持久化,子 Agent 与定时任务创建的会话进入各自的文件夹,每个分组按页加载而不再一次拉取全部列表。按 Agent 分组仍然只差一次切换。 + +模型下拉框现在优先列出真正配置了 Key 的模型,其余的收在下面一行。折叠后的侧边栏变成了完整的八项导航栏,带中英双语悬浮提示(Benchmark 此前完全缺失)。自建模型分组与 Agent 改用首字母头像——底色由名称派生,深浅色模式下都满足 WCAG AA 对比度——同名模型跨分组终于能分辨了。 + +聊天渲染也做了一轮:链接在新标签页打开;长 URL 与行内代码在容器边缘换行,同时不会把中文段落里的英文单词拦腰截断;宽表格在消息内部横向滚动而不是把页面撑宽;展开的子 Agent 对话渲染在工具调用自身输出的下方;三个此前在移动端最多溢出 143px 的下拉面板现在留在视口内。费用中心的每日 Token 提示气泡跟随指针并显示缓存命中率;复制消息时生成的任务统计行此前是硬编码中文,现在同样走词典。 + +### 技能 + +AI 应用开发分组新增三个技能——**vllm**、**ollama** 与 **llamafactory**——让 Agent 不只是调用模型,还能把自己依赖的模型部署起来、调优出来。两个部署技能共享同一套引导流程:先问要部署哪个模型、再问用户偏好哪个引擎,然后部署、验证,最后用 CLI 注册端点。`penguin-cli`(现为 v5)与 `penguin-sdk` 承载了它们所依赖的硬性规则:给 Penguin 自身配置模型用默认数据根目录,而为在开发的 AI 应用配置模型必须写进该应用自己的项目目录。我们另有一篇[实践文章](/blog/natural-language-training-loop)讲当 Agent 同时握有这三个技能之后会发生什么——你不再敲命令,只描述你要的结果。 + +第四个技能 **bento-slides** 进入办公效率分组:让 Agent 做演示文稿,它会真正产出一份 Bento 幻灯片——一个自包含的 `.bento.html`,文档本身就是 JSON——并把你的素材映射到图表、morph 转场与状态页,而不是堆一屏要点。它改编自 Bento 项目自己的 MIT 许可技能,并在正文中标注了出处。 + +`agenthub-models` 同步了 0.4.1:新的受支持模型注册表、客户端现在可能直接拒绝的配置参数,以及 Gemini 3.6 / Kimi K3 / GLM-5.2 系列及其推理强度旋钮。 + +### 站点、文档与工程 + +文档站与官网的导航栏现在是同一套布局——同样的容器宽度、同样的 Logo 区块、同样的右侧集群——此前它们已经漂移成两份近乎相同的实现。博客新增 **Tech practice** 分类、置顶文章,以及文章元信息(按语言格式化的日期、作者行、复制链接按钮)。两份 README 与官网首页也在人们真正会去找的位置列出了内置技能。 + +README 路线图新增两项——Agent company and templates,以及 company-level self evolving;`examples/` 下的自我进化示例重写成两个可运行脚本,让本地开源权重模型真正给自己打分、改自己的文件并重跑,而不再是一段固定的示意记录。加固方面,服务端现在校验分页与日期查询参数而不是照单全收,两个此前没有直接测试覆盖的内核模块补上了单元测试。 + +升级现在只需一条命令。`penguin update` 会解析最新版本,先把将要做的事讲清楚,再按你当初的安装方式原地升级——tarball 安装就重新执行官方安装脚本(保留原安装目录与是否内置运行时),npm/pnpm/yarn/bun 全局安装就用对应的包管理器。它拒绝改动源码检出,也从不碰你的数据目录。`penguin update --check` 只报告版本,不做任何修改。 + +## 获取方式 + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +penguin web +``` + +然后打开 Models 页面,填入 Gemini 或 OpenRouter 的 Key,选择 `gemini-3.6-flash` 即可。完整的发布说明见 [`changelog/0.1.1/`](https://github.com/Prism-Shadow/penguin-harness/blob/main/changelog/0.1.1/README.md)。 diff --git a/packages/landing/content/blog/natural-language-training-loop.en.md b/packages/landing/content/blog/natural-language-training-loop.en.md new file mode 100644 index 0000000..e604099 --- /dev/null +++ b/packages/landing/content/blog/natural-language-training-loop.en.md @@ -0,0 +1,143 @@ +--- +title: "Let the agent drive Ollama, vLLM and LlamaFactory: a data-safe self-closing training loop" +date: 2026-07-22 +category: practice +excerpt: PenguinHarness 0.1.1 ships the ollama, vllm and llamafactory skills. They are not three more command-line tools for you to learn — they are what the agent already knows, so "serve a model here, then fine-tune it until it passes my evaluation" becomes something you say rather than something you type. This post is mostly about why this harness drives those three tools well: what ships in the box, what it costs per request, and why the loop closes at all. The data in it never leaves your environment. +--- + +Three skills landed in 0.1.1: `ollama`, `vllm`, `llamafactory`. The easy misreading is that PenguinHarness now expects you to learn three more command-line tools. + +It is the opposite. The skills are not written for you — they are written for the agent, and the entire reason they exist is that you stop running the commands. You say what you want in a sentence; the agent picks the tool, asks the questions it is required to ask, runs the thing, checks that it worked, and tells you what happened. + +This post is about what that changes, in three parts: + +1. **The interface is a sentence.** What you type, and what the agent then does on its own. +2. **Why *this* harness drives these three tools well.** Six things that are in the box rather than in your head, each one a file you can open. +3. **The loop closes, and nothing you care about goes out.** Serve → evaluate → fine-tune → redeploy → measure again, on your hardware, with your weights. + +Everything attributed to the agent here is behavior a shipped skill actually specifies. The skills are plain Markdown and you can read them yourself under `packages/skills/skills/`. Where a skill stops and you take over, this post says so rather than pretending. + +## 1. "Run a local model on this machine — I don't want the data going anywhere" + +That sentence is the whole input. Here is what happens on the other side of it. + +**The agent loads the `ollama` skill and asks two questions before touching anything.** Not out of politeness: the skill forbids running any command until the goal is clear. It asks which model you want to run — and if you have no preference it proposes a small default, Qwen3.5-0.8B, with the reminder that the model has to fit the machine's RAM or VRAM. It asks which engine you prefer, because that choice is genuinely yours: Ollama is the simple default and the only option on macOS or a CPU-only box, while vLLM is for high-throughput GPU serving. + +**Then it looks before it leaps.** The skill's first rule is to check the current state — is Ollama installed, is something already serving — and if port 11434 already has an instance, to reuse it and never kill it. That is the same instinct 0.1.1 baked into the default system prompt: never kill a process you did not start; when a port is busy, take another one. + +**Then it does the boring parts you would have gotten wrong.** It installs Ollama if missing, pulls the model, and raises the context window — the step people skip and then spend an afternoon debugging, because Ollama's default window is small and agent sessions are not. The skill gives it both ways to do that: `OLLAMA_CONTEXT_LENGTH` in the server environment, or `num_ctx` baked into a model variant with a Modelfile. + +**Then it verifies, and only then registers.** A pulled model is invisible to PenguinHarness until it is added, and model configuration is the CLI's job: + +```bash +# what the agent ran — not a checklist for you +curl http://localhost:11434/v1/models + +penguin config model add --provider custom --client-type openai \ + --base-url http://localhost:11434/v1 --model-id qwen3.5:0.8b --api-key ollama +penguin config model list +``` + +The details in that command are decisions, not boilerplate, and the `penguin-cli` skill is where the agent learned them. A model in PenguinHarness is the `(provider, model_id)` pair and the group is **never** inferred from the id — gateways resell vendor models under their upstream ids, so a guess could send your key to somebody else's endpoint; `custom` is the group for any endpoint outside the built-in ones. `--client-type openai --base-url ` is the shape for any OpenAI chat-completion compatible server. And `--api-key ollama` is not decoration: Ollama accepts any key, but the field must be non-empty. + +If the local model has a small context window, the agent also has a reason to cap its output — `--max-tokens` is a per-model cap, new in 0.1.1, that overrides the Agent default of 32,000, which on its own does not fit in a 32k window alongside any prompt at all. + +You did not run any of that. You approved it — each tool call, as it came. That is the first of the three places a human is still genuinely in the loop, and it is there on purpose. + +## 2. Why this harness drives these three tools well + +Any capable agent with a shell can, in principle, run `vllm serve`. The interesting question is what it takes to get from *in principle* to a session that works the first time, and how much of that you had to supply. Six answers, each one a file in this repo rather than an adjective. + +**One: the knowledge ships in the box.** `packages/skills/skills/vllm/SKILL.md`, `.../ollama/SKILL.md` and `.../llamafactory/SKILL.md` are part of the Skill library, and a project's `default_agent` is created with the whole library installed — not fetched, not configured, not pasted in by you. So on a fresh install the agent already knows that vLLM has to be started with `--enable-auto-tool-choice` and a `--tool-call-parser` matched to the model family or every tool call comes back a `400`; that Ollama's default context window is too small for agent sessions, and both ways to raise it; that a LoRA adapter must never be merged into a quantized base. None of that is exotic — it is exactly the set of things you find out by losing an afternoon to them. + +
+Expand: some of what the three skills already know + +Straight from the shipped `SKILL.md` files — this is the class of detail that is already in the agent's reach before you say anything: + +- **vLLM, tool calling.** `vllm serve --enable-auto-tool-choice --tool-call-parser hermes`, with the parser chosen for the model family (`hermes` for Qwen, `llama3_json` for Llama). Without it: `400 "auto" tool choice requires --enable-auto-tool-choice and --tool-call-parser to be set` on every agent request. +- **vLLM, out of memory at startup.** Lower `--gpu-memory-utilization` or `--max-model-len`, or serve a quantized model. +- **Ollama, context.** `OLLAMA_CONTEXT_LENGTH=32768 ollama serve`, or `PARAMETER num_ctx 32768` in a Modelfile plus `ollama create`. +- **Ollama, an instance already running.** Port 11434 in use means reuse it — never kill a serving process you did not start. +- **LlamaFactory, merging.** Merge the adapter into the base weights for standalone serving, never into a quantized base. +- **Registering the result.** `--provider custom --client-type openai --base-url `; a served model is invisible to Penguin until it is added. + +
+ +**Two: knowing many tools costs almost nothing per request.** Skills have no dedicated tool and are not concatenated into the prompt. The system prompt template carries a `{{SKILL_METADATA}}` placeholder, and assembly replaces it with one line per installed skill — `` - `vllm` — Deploy and serve LLMs with vLLM behind an OpenAI-compatible endpoint… `` — while the body is read from disk with a shell command only when a task matches. In 0.1.1 that is fifteen skills whose metadata lines total about 2.5 KB, against skill bodies totalling over 100 KB that never enter a context window until they are needed. Deep knowledge of a tool you are not using this turn is not a tax you pay every request. + +**Three: it operates the real CLIs, because that is all it has.** A session's built-in tools are `exec_command`, `input_command`, `run_subagent`, `input_subagent` and one image tool. There is no read-file tool, no write-file tool, and no per-vendor integration: `exec_command`'s own description tells the model to read, write and edit files and run programs with the shell. That constraint is the feature here. `vllm serve`, `ollama pull` and `llamafactory-cli train` need no adapter written for them, the flags in the skills are the tools' real flags rather than whatever subset a wrapper exposed, and when vLLM adds a flag next month the fix is a Markdown edit, not a release of this harness. + +**Four: the loop closes because registration is part of the job.** Serving a model and *being able to use* it are two different facts, and the gap between them is where most "the agent set it up for me" demos quietly end. The `penguin-cli` skill closes it: it teaches the agent that an endpoint is invisible until `penguin config model add` registers it, and — the part that actually matters — *which* data root to register it into. `--root` must point at the app's own data directory (`--root ./penguin_data`, the same path the app hands `createAgent({ root })`) when the agent is building an app; the default root is for Penguin's own model. So the model the agent just served becomes a model it can then run on, deliberately, in the right place. That is the difference between a loop and a demo. + +**Five: measuring is a shipped capability, not an exercise for the reader.** `benchmark-design`, `agent-evaluation` and `agent-optimization` are in the same library, installed the same way. Executing is not improving; "fine-tune it until it passes" needs something that can say *passes*. Part 3 is what those three do. + +**Six: the two stories are one story.** The reason the agent can drive these tools at all is that it drives them by running commands on the machine that holds the data. There is no hosted control plane in the middle that would need to see your dataset in order to orchestrate the run. "Local" is not a feature bolted onto this design; it is the same fact as "it can use the real CLI". + +**And the honest version of all six.** None of this says another agent cannot do these things. Give any competent shell-using agent the same instructions and it will serve the model too. The difference is smaller and checkable than a superlative: with PenguinHarness those instructions are already installed, cost about 2.5 KB of prompt to have available, and include the registration step that makes the served model usable afterwards. Elsewhere, you are the one who has to know it — and know it again next session. + +## 3. "Now fine-tune it until it passes my evaluation" + +This is the sentence that makes the loop close, and it only works because of the word *evaluation*. There is no self-improving anything without a number, and a single run is not a number. + +```text + ┌──────────────────────────────────────────────┐ + │ │ +serve the model → run the benchmark → read the failing traces + (vllm/ollama) (benchmark-design + (session ids from + agent-evaluation) the scoreboard) + ↑ │ + │ ↓ + redeploy ← merge and export ← fine-tune on what it got wrong + (vllm) (llamafactory) (llamafactory) +``` + +**The agent builds the measurement first.** `benchmark-design` has it lay out a Benchmark: a set of Cases, each with a public statement the tested agent sees and a private rubric it must never see, plus the scoreboard the results land in. Each Case runs more than once by default, because one sample from a nondeterministic local model is not a measurement. `agent-evaluation` runs each of those Case runs in isolation and returns nothing but protocol metadata — a score, a cost, a duration, a session id — which is how the rubric stays out of the tested agent's context. + +**Which gives the agent something to read, not just a number.** Every evaluation records the `(provider, model_id)` pair that produced it, so the base model and its tuned successor land on the same scoreboard and compare directly. And every run carries its session id, so the agent can open the exact trace and see which step lost the point. That is the whole reason "fine-tune on what it got wrong" is a sentence with a referent. + +**Then it fine-tunes.** The `llamafactory` skill has it confirm four things before training — available GPU memory (LoRA needs far less than full fine-tuning), the base model, the dataset and its format, and the goal, with LoRA SFT as the usual starting point. It registers the dataset in `data/dataset_info.json` in alpaca or sharegpt form, writes a training config derived from the shipped `examples/train_lora/qwen3_lora_sft.yaml`, runs `llamafactory-cli train`, and tries the result interactively before trusting it. + +**Then it redeploys and measures again.** The adapter gets merged into the base weights and exported. vLLM serves the export directory directly; Ollama needs an import first. When the agent serves it, the `vllm` skill has already told it the flag that everyone forgets — the tool-calling pair from part 2. The tuned endpoint gets registered as its **own** model id rather than overwriting the base one, precisely so both stay on the scoreboard, and then the same Benchmark runs again. + +That is the loop, and none of the steps between "serve" and "measure again" required you to name a command. + +**One seam, stated plainly.** Turning failing traces into training examples is the step the skills do *not* prescribe. `llamafactory` asks where your dataset lives and what format it is in; it does not teach an agent to mine a trace into an SFT file. The agent can read the traces, and it can write the conversion script if you ask it to — but that is you directing it, not a skill driving it. Anyone telling you this part is already automatic is selling something. + +## 4. Where the human still stands + +Three places, and they are not accidents: + +- **Approving tool calls.** Every tool call is gated. In the SDK that gate is literally a callback, and omitting it denies everything — the default is refusal, not permission. +- **The judgment calls.** Which base model. Which engine. What "good enough" means. All three serving and tuning skills are written to *ask* rather than assume, and `benchmark-design` requires you to name the agent under test and the capability being measured before it will start. If you also want the agent itself improved rather than the weights, `agent-optimization` works from the same scoreboard — and it refuses to touch Agent State until a snapshot exists to roll back to. +- **The dataset seam** from the end of part 3. + +Everything else — which flags, which port, which parser, whether to reuse the running server, what to do about `400 … tools must not be an empty array` (upgrade to 0.1.1, which stopped sending an empty tool list) or an out-of-memory at vLLM startup — is in the skills, which means it is in the agent. + +## 5. Data never leaves your environment + +This is the reason the whole arrangement is worth the trouble, so it deserves precision rather than a slogan. + +**Local, in this setup:** + +- **The served model.** Ollama exposes its OpenAI-compatible API on `http://localhost:11434/v1`; vLLM on `http://localhost:8000/v1`. Both are on the machine. Every prompt, tool schema, tool result and completion in an agent session against them stays on the loopback interface. +- **The training.** LlamaFactory runs on your GPU. Your dataset sits under `data/` next to `data/dataset_info.json`; the adapter and the merged export land under `saves/`. No stage of `llamafactory-cli train` ships your examples anywhere. +- **The evaluation.** Cases, statements and rubrics are files in your project — a rubric lives at a path like `~/.penguin/data/default_project/agents/tool-router/benchmarks/tool-routing-v1/CASE-003-pick-the-cheaper-endpoint/rubric/README.md` — and the evaluator reads them from disk. The scoreboard is a YAML file next to them. +- **The configuration.** `penguin config model add` writes into a single hidden project config file. It is CLI-managed, never hand-edited, and it stays where you point it. + +**What does cross the network:** installs and weights, coming *in*. `ollama pull`, `pip install vllm`, cloning LlamaFactory, resolving a Hugging Face base model id — all of them download. None of them upload your data. It is worth being clear about the direction, because "local" is often quietly claimed for setups that phone home. + +**And the one thing that decides everything else:** the model that drives the agent. If the agent is running on a hosted API, then the conversation itself — your instructions, the file contents it reads, the tool output it summarizes — goes to that vendor, no matter how local the model it is tuning. The fully-local configuration is a deliberate choice you make: point the harness's own default model at the local endpoint too, with the same `penguin config model add ... --set-default`, and the loop runs end to end without a third party in it. It is a real trade-off — a small local model driving the whole loop is not the same proposition as a frontier model driving it — and it should be a decision, not an assumption. + +## What actually changed + +Before this release, PenguinHarness could talk to any OpenAI-compatible endpoint but had nothing to say about where that endpoint came from. Standing one up, measuring what it could do, and fixing what it could not were three different tools with three sets of conventions, and a human in between translating. + +Now they are one system, and the human's job moved. You describe the outcome. The agent serves the model, runs its own benchmark, reads its own failures, tunes, redeploys, and measures again — choosing each next step from the last result. You approve the calls, make the judgment calls, and read the scoreboard. + +On your hardware. On your weights. With your data staying where you put it. + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +penguin web +``` diff --git a/packages/landing/content/blog/natural-language-training-loop.zh.md b/packages/landing/content/blog/natural-language-training-loop.zh.md new file mode 100644 index 0000000..01e3b30 --- /dev/null +++ b/packages/landing/content/blog/natural-language-training-loop.zh.md @@ -0,0 +1,143 @@ +--- +title: 让 Agent 来操作 Ollama、vLLM、LlamaFactory,做数据安全的模型训练自闭环 +date: 2026-07-22 +category: practice +excerpt: PenguinHarness 0.1.1 新增了 ollama、vllm、llamafactory 三个技能。它们不是三个要你去学的命令行工具,而是 Agent 本来就会的东西——于是「在这里部署一个模型,再把它微调到能过我的评测」变成了你说的一句话,而不是你敲的一串命令。本文主要讲的是:为什么这套 Harness 能把这三个工具用好——什么是开箱就有的、每次请求要付多少代价、以及闭环凭什么合得上。而整个过程中的数据不出你自己的环境。 +--- + +0.1.1 带来了三个技能:`ollama`、`vllm`、`llamafactory`。最容易的误读是:PenguinHarness 现在要你再学三个命令行工具。 + +恰恰相反。这些技能不是写给你的,是写给 Agent 的;它们存在的全部理由,就是让你不再敲这些命令。你用一句话说清楚要什么,Agent 自己选工具、问该问的问题、执行、验证,然后把结果讲给你听。 + +本文分三部分: + +1. **接口是一句话。** 你输入什么,Agent 随后自己做了什么。 +2. **为什么这套 Harness 能把这三个工具用得更好。** 六件写在仓库文件里、而不是留在你脑子里的事。 +3. **闭环合得上,而且数据不出域。** 部署 → 评测 → 微调 → 重新部署 → 再测一次,跑在你自己的硬件与权重上。 + +下文中归给 Agent 的每一项行为,都是已发布技能里真实写明的规则。这些技能就是普通的 Markdown,你可以自己去 `packages/skills/skills/` 下读。哪里是技能的边界、需要你接手,本文照实说明,不粉饰。 + +## 一、「在这台机器上跑一个本地模型,数据不要出去」 + +这句话就是全部输入。以下是它另一端发生的事。 + +**Agent 加载 `ollama` 技能,并在动手之前先问两个问题。** 这不是客气:技能明确规定目标不清楚之前不许执行任何命令。它问你要跑哪个模型——你没有偏好时,它会推荐一个小默认值 Qwen3.5-0.8B,并提醒模型必须装得进机器的内存或显存。它还问你偏好哪个引擎,因为这个选择确实属于你:Ollama 是简单默认项,也是 macOS 与纯 CPU 机器上的唯一选择;vLLM 面向高吞吐的 GPU 部署。 + +**然后它先看、再动。** 技能的第一条规则就是先确认当前状态——Ollama 装了没有、是不是已经在提供服务;如果 11434 端口上已经有实例,就复用它,绝不杀掉。这也正是 0.1.1 写进默认系统提示词的那条:绝不杀死不是自己启动的进程,端口被占就另选一个。 + +**然后它把你多半会做错的那几步做对。** 缺了就装 Ollama,拉取模型,并调大上下文窗口——这一步最常被跳过,然后花一个下午排查,因为 Ollama 默认的窗口很小,而 Agent 会话不小。技能给了它两条路:在服务环境里设 `OLLAMA_CONTEXT_LENGTH`,或者用 Modelfile 把 `num_ctx` 固化成一个模型变体。 + +**然后它先验证,再注册。** 拉下来的模型在被添加之前,对 PenguinHarness 是不可见的,而模型配置是 CLI 的职责: + +```bash +# 这是 Agent 执行过的命令,不是给你照做的清单 +curl http://localhost:11434/v1/models + +penguin config model add --provider custom --client-type openai \ + --base-url http://localhost:11434/v1 --model-id qwen3.5:0.8b --api-key ollama +penguin config model list +``` + +这条命令里的每个细节都是判断,不是模板,而 Agent 是从 `penguin-cli` 技能里学到它们的。PenguinHarness 里的模型由 `(provider, model_id)` 二元组确定,分组**永远不**从模型 ID 推断——网关会用上游 ID 转售厂商模型,猜错就可能把你的 Key 发到别人的端点上;`custom` 是内置分组之外任何端点所属的分组。`--client-type openai --base-url ` 是所有 OpenAI chat-completion 兼容服务的写法。而 `--api-key ollama` 不是装饰:Ollama 接受任何 Key,但这个字段必须非空。 + +如果本地模型的上下文窗口很小,Agent 还有理由顺手限制输出——`--max-tokens` 是 0.1.1 新增的模型级上限,优先级高于 Agent 默认的 32,000;而这个默认值本身就塞不进 32k 窗口,更别提再加上提示词。 + +以上没有一条是你跑的。你做的是逐条批准工具调用。这是人仍然实实在在留在环里的第一处,而且是有意为之。 + +## 二、为什么这套 Harness 能把这三个工具用好 + +任何一个有 Shell 的能干 Agent,原则上都跑得动 `vllm serve`。真正的问题是:从「原则上可以」到「这一次会话第一遍就跑通」,中间要补多少东西,而其中又有多少得由你来补。下面六条回答,每一条对应仓库里的一个文件,而不是一个形容词。 + +**其一:这些知识是开箱就有的。** `packages/skills/skills/vllm/SKILL.md`、`.../ollama/SKILL.md`、`.../llamafactory/SKILL.md` 本身就属于技能库,而一个项目的 `default_agent` 在创建时会把整个技能库装上——不用去拉取、不用去配置,也不用你粘贴进去。所以在一台全新安装的机器上,Agent 已经知道:vLLM 必须带上 `--enable-auto-tool-choice` 和与模型家族匹配的 `--tool-call-parser` 启动,否则每一次工具调用都会返回 `400`;Ollama 的默认上下文窗口撑不住 Agent 会话,以及两种调大它的办法;LoRA 适配器绝不能合并到量化过的基座上。这些都不算冷门知识——它们恰好就是那种「搭进去一个下午才会知道」的知识。 + +
+展开:三个技能里已经写着的一部分坑 + +以下直接来自已发布的 `SKILL.md`——这就是在你开口之前,Agent 就已经能够拿到的细节: + +- **vLLM 的工具调用。** `vllm serve --enable-auto-tool-choice --tool-call-parser hermes`,parser 按模型家族选(Qwen 用 `hermes`,Llama 用 `llama3_json`)。不带它,每一次 Agent 请求都会得到 `400 "auto" tool choice requires --enable-auto-tool-choice and --tool-call-parser to be set`。 +- **vLLM 启动时显存不足。** 调低 `--gpu-memory-utilization` 或 `--max-model-len`,或者改部署量化模型。 +- **Ollama 的上下文。** `OLLAMA_CONTEXT_LENGTH=32768 ollama serve`,或在 Modelfile 里写 `PARAMETER num_ctx 32768` 再 `ollama create`。 +- **Ollama 已经在跑。** 11434 端口被占就复用——绝不杀死一个不是自己启动的服务进程。 +- **LlamaFactory 的合并。** 为了能独立部署要把适配器合并进基座权重,但绝不合并到量化过的基座上。 +- **把结果注册回去。** `--provider custom --client-type openai --base-url `;模型部署起来之后,不添加就对 Penguin 不可见。 + +
+ +**其二:会用很多工具,几乎不增加每次请求的开销。** 技能没有专门的工具,也不会被整段拼进提示词。系统提示词模板里只有一个 `{{SKILL_METADATA}}` 占位符,装配时把它替换成每个已安装技能一行的元数据——形如 `` - `vllm` — Deploy and serve LLMs with vLLM behind an OpenAI-compatible endpoint… ``——而技能正文要等任务真的对上了,才由模型用 Shell 命令从磁盘读进来。在 0.1.1 里,这是十五个技能、元数据合计约 2.5 KB,而它们的正文合计超过 100 KB,在被需要之前一个字都不进上下文窗口。对一个这一轮用不上的工具了解得很深,并不需要你每次请求都为它付费。 + +**其三:它操作的就是真正的 CLI,因为它手里只有这个。** 一次会话的内置工具是 `exec_command`、`input_command`、`run_subagent`、`input_subagent`,外加一个图像工具。没有读文件工具,没有写文件工具,也没有任何按厂商定制的集成:`exec_command` 自己的描述就是让模型用 Shell 去读写编辑文件、运行程序。这个约束在这里恰恰是优点。`vllm serve`、`ollama pull`、`llamafactory-cli train` 不需要谁去写一层适配,技能里写的参数就是工具真正的参数,而不是某层封装挑出来的子集;等 vLLM 下个月加了新参数,要改的是一份 Markdown,而不是发一个 Harness 的新版本。 + +**其四:闭环合得上,是因为「注册」本身就是工作的一部分。** 把模型部署起来,和**能用上**这个模型,是两件事;而大多数「Agent 帮我搭好了」的演示,正是在这条缝上悄悄结束的。`penguin-cli` 技能把它补上了:它告诉 Agent,端点在 `penguin config model add` 注册之前是不可见的;更关键的是,要注册到**哪个**数据根目录里。Agent 在开发一个应用时,`--root` 必须指向该应用项目内自己的数据目录(`--root ./penguin_data`,也就是应用传给 `createAgent({ root })` 的那个路径),而默认根目录是留给 Penguin 自身的模型的。于是它刚刚部署好的模型,就成了它接下来可以运行其上的模型——有意为之,且落在正确的位置。演示和闭环的差别就在这里。 + +**其五:「能衡量」是随包发布的能力,不是留给读者的练习。** `benchmark-design`、`agent-evaluation`、`agent-optimization` 在同一个技能库里,用同样的方式安装。执行不等于改进;「微调到能过」这句话,得先有个东西能说出「过了」。第三部分讲的就是这三个技能做了什么。 + +**其六:这两个故事其实是同一个故事。** Agent 之所以能驱动这些工具,正是因为它是在持有数据的那台机器上直接执行命令。中间没有一个托管的控制面,需要先看到你的数据集才能编排这次运行。「本地」不是给这套设计外挂上去的特性,它和「它能用真正的 CLI」是同一件事。 + +**以及这六条的诚实版本。** 以上没有一条是在说别的 Agent 做不到。把同样的说明喂给任何一个会用 Shell 的强 Agent,它也能把模型部署起来。差别比一句「最强」小得多,也更可核查:在 PenguinHarness 上,这些说明是已经装好的,代价是约 2.5 KB 的提示词,而且其中包含了那一步让模型在部署之后真正可用的注册。换到别处,得由你来知道这些——并且下一次会话还得再知道一遍。 + +## 三、「把这个模型微调到能过我的评测」 + +正是这句话让闭环合上,而关键在「评测」两个字。没有数字就没有自我改进,而跑一次不算数字。 + +```text + ┌───────────────────────────────────────────────┐ + │ │ + 部署模型 → 跑评测 → 读失败的 Trace + (vllm/ollama) (benchmark-design + (记分板里的 session id) + agent-evaluation) + ↑ │ + │ ↓ + 重新部署 ← 合并并导出 ← 针对失分微调 + (vllm) (llamafactory) (llamafactory) +``` + +**Agent 先把量具造出来。** `benchmark-design` 让它铺开一个 Benchmark:一组 Case,每个 Case 都有被测 Agent 能看到的公开 Statement 和它绝不能看到的私有 Rubric,以及结果落地的记分板。每个 Case 默认不止跑一次——对一个不确定的本地模型采样一次,算不上一次测量。`agent-evaluation` 负责隔离地执行每一次 Case 运行,且只返回协议元数据:分数、成本、耗时、session id。正是这份沉默让 Rubric 不进入被测 Agent 的上下文。 + +**于是 Agent 拿到的不只是一个数字。** 每次评估都记录产生它的 `(provider, model_id)` 二元组,基座模型与调优后的后继者因此落在同一张记分板上,可以直接比。而每一次 run 都带着自己的 session id,Agent 可以打开那次 Trace,看清是哪一步丢的分。「针对失分微调」这句话之所以有确切的所指,全靠这一点。 + +**然后它开始微调。** `llamafactory` 技能要求它在训练前确认四件事:可用显存(LoRA 的需求远低于全参微调)、基座模型、数据集及其格式、以及目标——常规起点是 LoRA SFT。它把数据集按 alpaca 或 sharegpt 格式登记进 `data/dataset_info.json`,从随附的 `examples/train_lora/qwen3_lora_sft.yaml` 派生出训练配置,执行 `llamafactory-cli train`,并在信任结果之前先交互试一下。 + +**然后它重新部署,再测一次。** 适配器被合并进基座权重并导出。vLLM 可以直接部署导出目录,Ollama 则需要先导入。部署时,`vllm` 技能早就告诉过它那个所有人都会忘的参数——就是第二部分里那对工具调用开关。调优后的端点被注册成**独立**的模型 ID,而不是覆盖基座——正是为了让两者都留在记分板上——然后同一个 Benchmark 再跑一遍。 + +这就是闭环,而从「部署」到「再测一次」之间的每一步,都不需要你说出任何一条命令。 + +**有一道缝,明说。** 把失败的 Trace 变成训练样本,是技能**没有**规定的一步。`llamafactory` 只问你数据集在哪、是什么格式,它并不教 Agent 如何从 Trace 里挖出一份 SFT 数据。Agent 能读 Trace,你让它写转换脚本它也能写——但那是你在指挥它,不是技能在驱动它。谁要是告诉你这一段已经全自动了,那多半是在卖东西。 + +## 四、人还站在哪里 + +三个位置,都不是疏漏: + +- **批准工具调用。** 每一次工具调用都要过闸。在 SDK 里这个闸就是一个回调函数,不提供它等于全部拒绝——默认是拒绝,而不是放行。 +- **只能由你做的判断。** 用哪个基座模型、用哪个引擎、多好才算「够好」。三个部署与调优技能都写成**先问**而不是替你假设;`benchmark-design` 也要求你先指明被测 Agent 与要衡量的能力才肯开始。如果你想改进的是 Agent 本身而不是权重,`agent-optimization` 基于同一张记分板工作——而且在存在可回滚的快照之前,它拒绝改动 Agent State。 +- **数据集那道缝**,见第三部分结尾。 + +除此之外的一切——用哪些参数、哪个端口、哪个 parser、要不要复用已在运行的服务、遇到 `400 … tools must not be an empty array` 怎么办(升级到 0.1.1,它不再发送空工具列表)、vLLM 启动显存不足怎么办——都写在技能里,也就意味着都在 Agent 身上。 + +## 五、数据安全不出域 + +这才是整套安排值得折腾的理由,所以它需要的是精确,而不是口号。 + +**这套配置下留在本地的部分:** + +- **被部署的模型。** Ollama 在 `http://localhost:11434/v1` 暴露 OpenAI 兼容 API,vLLM 在 `http://localhost:8000/v1`。两者都在本机。对它们发起的 Agent 会话里,每一条提示词、工具 Schema、工具结果与补全,都留在本地回环接口上。 +- **训练。** LlamaFactory 跑在你自己的 GPU 上。数据集放在 `data/` 下、与 `data/dataset_info.json` 相邻;适配器与合并后的导出落在 `saves/`。`llamafactory-cli train` 的任何阶段都不会把你的样本发出去。 +- **评测。** Case、Statement 与 Rubric 都是你项目里的文件——一份 Rubric 的路径形如 `~/.penguin/data/default_project/agents/tool-router/benchmarks/tool-routing-v1/CASE-003-pick-the-cheaper-endpoint/rubric/README.md`——评估者从磁盘读取它们,记分板是与之相邻的一份 YAML。 +- **配置。** `penguin config model add` 写入一份隐藏的项目配置文件。它只由 CLI 管理、从不手工编辑,而且你指到哪它就待在哪。 + +**确实会经过网络的部分:** 安装包与权重,方向是**进来**。`ollama pull`、`pip install vllm`、克隆 LlamaFactory、解析一个 Hugging Face 基座模型 ID——这些都是下载,没有一项在上传你的数据。方向值得说清楚,因为「本地」这个词经常被悄悄安在会回传的方案上。 + +**以及决定其余一切的那一项:** 驱动 Agent 的模型本身。如果 Agent 跑在托管 API 上,那么无论它调优的模型多本地,对话本身——你的指令、它读到的文件内容、它转述的工具输出——都会到达那家厂商。完全本地是你主动做的选择:用同样的 `penguin config model add ... --set-default`,把 Harness 自身的默认模型也指向本地端点,整个闭环就没有第三方参与。这是一个真实的取舍——让一个小的本地模型驱动整个闭环,和让一个前沿模型驱动它,不是同一件事——它应该是一个决定,而不是一个默认假设。 + +## 真正改变的是什么 + +在这个版本之前,PenguinHarness 可以对接任何 OpenAI 兼容端点,但对这个端点从哪来只字不提。把模型部署起来、衡量它能做到什么、修补它做不到的地方,是三套工具、三套约定,中间夹着一个人做翻译。 + +现在它们是同一套系统,而人的位置变了。你描述你要的结果。Agent 自己部署模型、自己跑评测、自己读失败原因、自己微调、自己重新部署、自己再测一次——每一步都由上一步的结果决定。你负责批准调用、做判断、看记分板。 + +跑在你自己的硬件上。用你自己的权重。数据留在你放它的地方。 + +```bash +curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh +penguin web +``` diff --git a/packages/landing/package.json b/packages/landing/package.json index a3be8ef..d0f9a0e 100644 --- a/packages/landing/package.json +++ b/packages/landing/package.json @@ -1,6 +1,6 @@ { "name": "@prismshadow/penguin-landing", - "version": "0.1.0", + "version": "0.1.1", "private": true, "type": "module", "description": "PenguinHarness landing page (React + Vite + Tailwind CSS): multilingual, light/dark themes, benchmark showcase, and a local Markdown blog, deployed to GitHub Pages via GitHub Actions.", @@ -17,6 +17,7 @@ "react-dom": "^19.1.0", "react-markdown": "^10.1.0", "react-router": "^7.6.0", + "rehype-raw": "^7.0.0", "remark-gfm": "^4.0.0" }, "devDependencies": { diff --git a/packages/landing/public/blog-assets/gemini-3-6-flash-evals.webp b/packages/landing/public/blog-assets/gemini-3-6-flash-evals.webp new file mode 100644 index 0000000..0dcbce1 Binary files /dev/null and b/packages/landing/public/blog-assets/gemini-3-6-flash-evals.webp differ diff --git a/packages/landing/src/components/announcement-bar.tsx b/packages/landing/src/components/announcement-bar.tsx index 3765ad2..04bf2db 100644 --- a/packages/landing/src/components/announcement-bar.tsx +++ b/packages/landing/src/components/announcement-bar.tsx @@ -14,8 +14,13 @@ import { ArrowRightIcon } from "./icons"; const ROTATE_MS = 6000; -/** Both announcements point at blog posts (content/blog/.*.md). */ +/** + * Every announcement points at a blog post (content/blog/.*.md), newest + * first: slide 0 is what a visitor sees before the rotation moves, so the + * freshest news goes at the front. + */ const ITEMS = [ + { key: "gemini", to: "/blog/gemini-3-6-in-penguinharness" }, { key: "models", to: "/blog/introducing-penguinharness" }, { key: "fireworks", to: "/blog/fireworks-credits-amd" }, ] as const; @@ -28,6 +33,7 @@ function prefersReducedMotion(): boolean { export function AnnouncementBar() { const texts: Record<(typeof ITEMS)[number]["key"], string> = { + gemini: S.announcement.gemini, models: S.announcement.models, fireworks: S.announcement.fireworks, }; diff --git a/packages/landing/src/lib/strings-en.ts b/packages/landing/src/lib/strings-en.ts index 0a4d9c3..67d2ded 100644 --- a/packages/landing/src/lib/strings-en.ts +++ b/packages/landing/src/lib/strings-en.ts @@ -12,6 +12,7 @@ export const en: Strings = { label: "Announcements", prev: "Previous announcement", next: "Next announcement", + gemini: "Gemini 3.6 Flash is now available in PenguinHarness", models: "Kimi K3 and Qwen 3.8 Max are now available in PenguinHarness", fireworks: "Claim $50 in Fireworks API credits with the AMD Developer Program", }, diff --git a/packages/landing/src/lib/strings.ts b/packages/landing/src/lib/strings.ts index 209ad21..4094a46 100644 --- a/packages/landing/src/lib/strings.ts +++ b/packages/landing/src/lib/strings.ts @@ -13,6 +13,7 @@ export const zh = { label: "公告", prev: "上一条公告", next: "下一条公告", + gemini: "Gemini 3.6 Flash 现已在 PenguinHarness 可用", models: "Kimi K3 与 Qwen 3.8 Max 模型现已在 PenguinHarness 可用", fireworks: "携手 AMD 开发者计划:$50 Fireworks API 额度免费领取中", }, diff --git a/packages/landing/src/pages/blog-list.tsx b/packages/landing/src/pages/blog-list.tsx index 495cfd8..4d692c5 100644 --- a/packages/landing/src/pages/blog-list.tsx +++ b/packages/landing/src/pages/blog-list.tsx @@ -56,7 +56,10 @@ export function BlogListPage() {