From 9767fe60e9e048303e2b0a30d3bfa510a23ad601 Mon Sep 17 00:00:00 2001 From: Yaowei Zheng Date: Thu, 23 Jul 2026 02:06:43 +0800 Subject: [PATCH] docs(changelog): 0.2.0 entries for the July 22 batch (#30) Co-authored-by: Alice Co-authored-by: Claude Fable 5 --- README.md | 2 +- README.zh.md | 2 +- .../0.2.0/2026-07-22-docs-and-examples.md | 4 ++ changelog/0.2.0/2026-07-22-models-and-core.md | 44 +++++++++++++++++++ changelog/0.2.0/2026-07-22-sites-and-blog.md | 13 ++++++ changelog/0.2.0/2026-07-22-skills.md | 13 ++++++ changelog/0.2.0/2026-07-22-tooling.md | 4 ++ changelog/0.2.0/2026-07-22-web-app.md | 35 +++++++++++++++ changelog/0.2.0/README.md | 15 +++++++ packages/landing/src/lib/strings-en.ts | 5 ++- packages/landing/src/lib/strings.ts | 5 ++- 11 files changed, 138 insertions(+), 4 deletions(-) create mode 100644 changelog/0.2.0/2026-07-22-docs-and-examples.md create mode 100644 changelog/0.2.0/2026-07-22-models-and-core.md create mode 100644 changelog/0.2.0/2026-07-22-sites-and-blog.md create mode 100644 changelog/0.2.0/2026-07-22-skills.md create mode 100644 changelog/0.2.0/2026-07-22-tooling.md create mode 100644 changelog/0.2.0/2026-07-22-web-app.md create mode 100644 changelog/0.2.0/README.md diff --git a/README.md b/README.md index d85cbde..63aa7ed 100644 --- a/README.md +++ b/README.md @@ -74,7 +74,7 @@ Four Skill groups ship in the box ([docs](https://penguin.ooo/docs/skills)); Age | -------------------- | --------------------------------------------------------------------------------- | | Office Productivity | `data-analysis`, `firecrawl` | | Software Development | `web-design`, `software-engineering` | -| AI App Development | `penguin-sdk`, `penguin-cli`, `agenthub-models` | +| AI App Development | `penguin-sdk`, `penguin-cli`, `agenthub-models`, `vllm`, `ollama`, `llamafactory` | | Agent Tuning | `agent-creation`, `benchmark-design`, `agent-evaluation`, `agent-optimization` | ## Supported Models diff --git a/README.zh.md b/README.zh.md index d2c174b..c27d072 100644 --- a/README.zh.md +++ b/README.zh.md @@ -74,7 +74,7 @@ https://github.com/user-attachments/assets/aec49ae9-b743-467b-b247-37bedfeaa36e | ----------- | ------------------------------------------------------------------------------ | | 办公效率 | `data-analysis`、`firecrawl` | | 软件开发 | `web-design`、`software-engineering` | -| AI 应用开发 | `penguin-sdk`、`penguin-cli`、`agenthub-models` | +| AI 应用开发 | `penguin-sdk`、`penguin-cli`、`agenthub-models`、`vllm`、`ollama`、`llamafactory` | | Agent 调优 | `agent-creation`、`benchmark-design`、`agent-evaluation`、`agent-optimization` | ## 支持的模型 diff --git a/changelog/0.2.0/2026-07-22-docs-and-examples.md b/changelog/0.2.0/2026-07-22-docs-and-examples.md new file mode 100644 index 0000000..ae992f8 --- /dev/null +++ b/changelog/0.2.0/2026-07-22-docs-and-examples.md @@ -0,0 +1,4 @@ +# Docs and examples: roadmap items and a genuinely self-evolving demo + +- The README roadmap (both languages) gains two items: Agent company and templates, and company-level self evolving. +- The self-improvement example under `examples/` was reworked to be genuinely self-evolving: the demo now ships two runnable scripts that let a local open-weight model score itself, edit its own files and re-run — the loop the landing page demonstrates — instead of a fixed illustrative transcript. diff --git a/changelog/0.2.0/2026-07-22-models-and-core.md b/changelog/0.2.0/2026-07-22-models-and-core.md new file mode 100644 index 0000000..955949f --- /dev/null +++ b/changelog/0.2.0/2026-07-22-models-and-core.md @@ -0,0 +1,44 @@ +# Models and core: request hygiene, prompt guardrails, and runtime settings + +## Empty tool lists stay off the wire + +Strict OpenAI-compatible servers reject requests that carry `tools: []` — vLLM answers `400 … tools must not be an empty array. Either provide at least one tool or omit the field entirely.` Every tool-less request the harness makes (the Models-page connectivity probe, session-title generation, the vision describer) hit this against a local vLLM endpoint. + +- `buildUniConfig` now omits the `tools` field entirely when the tool list is empty, instead of sending an empty array. Tool-carrying agent requests are unchanged. +- `tool_choice` was investigated end to end: neither this repo nor AgentHub 0.4.0 ever sends it. The `400 "auto" tool choice requires --enable-auto-tool-choice and --tool-call-parser` failure seen on vLLM is produced server-side when a non-empty `tools` array arrives without those flags — real tool use on vLLM needs them regardless of client (the vllm skill documents this). A wire-level capture test now locks both behaviors: no `tools` key and no `tool_choice` key on a tool-less request. + +## Reserved-port and API-key guardrails in the default system prompt + +Agents occasionally freed a busy port by killing its listener — sometimes the harness's own services. The default server port now has a single source: core's internal `ports.ts` exports `DEFAULT_SERVER_PORT`, narrowly re-exported from the package barrel for the CLI and server to derive their defaults from (still runtime-overridable via `--port` / `PORT`). The prompt rule deliberately carries no hardcoded numbers: never kill a process you did not start — including PenguinHarness's own services — never take a harness service port for your own servers, and when a wanted port is busy, pick another free port instead of killing the listener. + +On an API authentication/authorization or API-key error (401/403, invalid or missing key), the agent retries at most once; if the error persists it stops calling tools and asks the user to update the key in the agent's vault or the model settings outside the chat — secret values never belong in the conversation. Updated secrets only take effect in the next conversation, so further retries cannot succeed — the prompt says so. + +The default prompt is a seed for each agent's editable `system_config.yaml`: existing agents keep their current prompt; new agents get the rules. + +## Max output tokens is a per-model setting + +A model with a 32k context served locally rejected requests outright — `400 This model's maximum context length is 32768 tokens. However, you requested 32000 output tokens…` — and took session-title generation down with it. The Models page (and `penguin config model add --max-tokens`) now accepts a per-model max output tokens cap, stored on the model entry and applied ahead of the agent-level default; out-of-band requests (title generation, vision description) respect it too, taking the smaller of their own cap and the model's. Unset means today's behavior. + +## Thinking level moves to the conversation + +The thinking level is no longer a Models-page annotation. The default lives in Agent settings (where it always was), and the chat draft gains a compact titled picker next to the model selector offering `low` / `medium` / `high` / `xhigh` — `none` is no longer offered because many models cannot disable thinking, though it remains a valid stored value that still displays correctly. Changing the picker writes through to the Agent settings immediately, so the session created on send — and every later one — uses the new level. The draft's model choice now carries over the same way: after a successful send it becomes the next conversation's default. + +## Subagents follow the parent session + +`run_subagent` with the model pair omitted used to fall through to the Project default model, and a child always ran at its own Agent's thinking level. A spawned subagent now inherits the parent session's resolved `(provider, model_id)` pair and its effective thinking level (workspace was already inherited); an explicit complete pair in the tool call still wins, and half a pair is still rejected rather than completed from the parent's half. The pass-down is tri-state, so a parent with no thinking level produces a child with none — the child's own config never sneaks back in — and resuming a session restores the thinking level its Trace recorded instead of re-reading the Agent config. + +## Sessions carry their origin + +`session_meta` gains an optional `source` field (`"subagent" | "schedule"`, absent = user-created) written at creation, preserved across resume and compaction-driven trace rotation, and treated as the single source of truth: the server derives the session index's origin from the meta (registering children from the forwarded meta, adopting discovered traces, lazily reading the trace head for rows indexed by an earlier process) and no longer stores the type in the database. + +## AgentHub 0.4.1 and a catalog refresh across every provider group + +The SDK dependency moves to AgentHub 0.4.1 (core and CLI). The upgrade is type-compatible — the published `UniConfig`, `ThinkingLevel`, message/event types and `AutoLLMClient` declarations are unchanged from 0.4.0, so nothing in core needed adapting. What 0.4.1 adds is additive: a `listSupportedModels(currency)` registry (model / base URL / client triples with modalities, context windows and per-million pricing), a typed `UnsupportedParameterError` for rejected `temperature` / `tool_choice` / `prompt_caching` values, and client support for the Gemini 3.6 generation, Kimi K3 and GLM-5.2. PenguinHarness never sends those three parameters, so the new error cannot fire from here today. + +That registry doubles as the authoritative model list, so the catalog was diffed against it and refreshed everywhere it fell behind — 59 entries to 70. New rows: Gemini 3.6 Flash and Gemini 3.5 Flash Lite on both the Google endpoint and OpenRouter; Claude Fable 5 and Claude Sonnet 5 on Anthropic; Kimi K3 on Moonshot; Kimi K2.6, Qwen3.6 35B A3B and GLM 5.1 on OpenRouter; and Kimi K2.6, GLM 5.1 and Qwen3.6 35B A3B on SiliconFlow — the last three unpriced, because no source publishes their rates and a guessed number is worse than none. Every context window, vision flag and price bucket came from the registry rather than a vendor page. `google/gemini-3.5-flash`'s context window was 1,000,000 against the registry's and the direct-vendor row's 1,048,576, and is now corrected. Adding Kimi K3 also surfaced a routing gap: `resolveModelEnv` matched only `kimi-k2.x`, so the new id resolved to no credential pair until a `kimi-k3` branch was added. + +The READMEs' model tables move the other way — one row per vendor family naming only the newest generation, with the full preset list left to the app's Models page. Kimi K2.6 and Gemini 3.5 Flash drop out; Claude Opus 4.8 becomes Claude 5 and GPT 5.5 becomes GPT 5.6 under the same rule. + +## Session-title generation is internal + +`session-title.ts` moved into core's `internal/` module. `Session.generateTitle()` remains the public entry point, and `SessionTitleResult`, `stripConversationMarkers` and `sanitizeTitle` (both used by the server's fallback title path) stay importable from the package barrel; the LLM-driving internals (`buildTitlePrompt`, `generateTitleWithLLM`) are no longer part of the public surface. diff --git a/changelog/0.2.0/2026-07-22-sites-and-blog.md b/changelog/0.2.0/2026-07-22-sites-and-blog.md new file mode 100644 index 0000000..c56d828 --- /dev/null +++ b/changelog/0.2.0/2026-07-22-sites-and-blog.md @@ -0,0 +1,13 @@ +# Sites: one navbar, a richer blog, and the built-in Skills listed + +## The docs and landing navbars are identical + +The two sites' navbars differed in container width (6xl vs 7xl), the docs-only badge pill, hamburger placement and a broken menu animation class. Both now share the same `max-w-7xl` container (the landing footer aligned to match, framing nav and footer consistently while content sections stay 6xl), the same logo block, and the same right-cluster layout; the landing language menu's undefined `anim-pop` class was replaced with the working `anim-fade`. Cross-SPA link semantics and each site's mobile behavior stay as they were. + +## Blog categories, pinned posts, and page metadata + +The blog list stays a single flat list with category badges and filter chips, now across three categories — Product news, Release notes, and the new Tech practice, which the AMD local-agents post moved into. Posts can be pinned to the top via `pinned: true` frontmatter; the launch post introducing PenguinHarness is pinned. A second practice post joined the blog: implementing agent self-improvement with PenguinHarness on an AMD GPU (en + zh), adopted into the same category and author conventions. The detail page moves its metadata below the title: a locale-formatted date ("July 20, 2026" / "2026年7月20日"), the author line (frontmatter `author`, defaulting to Yaowei Zheng (PrismShadow AI)), and a copy-page-link button with a safe clipboard fallback and a transient "Copied" state. + +## The built-in Skills, listed where people look + +The READMEs (both languages) gain a compact Built-in Skills section — one table of the four skill groups and their members — and the landing page gains a matching Skills section of group cards between Features and Security. The lists cover what currently ships and grow as new skills land — refreshed in this release to include the vLLM/Ollama serving and LlamaFactory fine-tuning skills once they landed. diff --git a/changelog/0.2.0/2026-07-22-skills.md b/changelog/0.2.0/2026-07-22-skills.md new file mode 100644 index 0000000..217248a --- /dev/null +++ b/changelog/0.2.0/2026-07-22-skills.md @@ -0,0 +1,13 @@ +# Skills: local serving and fine-tuning join AI App Development + +Three new built-in skills extend the AI App Development group so agents can stand up and tune the models they build on: + +- **vllm** — install and serve models with vLLM's OpenAI-compatible server, including the tool-calling flags agent harnesses need (`--enable-auto-tool-choice --tool-call-parser …`), quantization and memory options, and troubleshooting. +- **ollama** — pull and serve local models with Ollama, its OpenAI-compatible endpoint, and context-length tuning via `OLLAMA_CONTEXT_LENGTH` or a Modelfile. +- **llamafactory** — fine-tune with LlamaFactory (dataset registration, `llamafactory-cli train/export/chat/api` driven by the upstream example YAMLs), then serve the result with vLLM or with Ollama via a Modelfile import. + +Both serving skills share a guided workflow: ask the user which model to serve (recommending the small Qwen3.5-0.8B default when they have no preference) and which engine they prefer — Ollama as the approachable default, vLLM when throughput matters — then serve, verify, and register the endpoint with the penguin CLI (models appear in `penguin config model list` only after `model add`). Each skill's icon follows the project's official mark. + +The hard root-separation guardrail lives where configuration is taught — the `penguin-cli` skill (now v5) and the `penguin-sdk` skill: configuring Penguin's own model uses the default root, but models configured for an AI app under development must use the app's own project directory (`--root ./penguin_data`) unless the user chose otherwise — never the global `~/.penguin/data`. The serving skills point at that rule rather than restating it. + +The `agenthub-models` skill tracks the AgentHub 0.4.1 upgrade: it documents the new supported-model registry, the config parameters a client may now reject outright, and the Gemini 3.6 / Kimi K3 / GLM-5.2 families with their reasoning-effort knobs, alongside a routing table rewritten from the release's actual client-matching order. diff --git a/changelog/0.2.0/2026-07-22-tooling.md b/changelog/0.2.0/2026-07-22-tooling.md new file mode 100644 index 0000000..499e389 --- /dev/null +++ b/changelog/0.2.0/2026-07-22-tooling.md @@ -0,0 +1,4 @@ +# Tooling and hardening: request validation and core test coverage + +- The server now validates `positiveIntParam` and `optionalDateParam` inputs instead of trusting query strings — malformed paging/date parameters return a clean 400 rather than leaking into SQL or arithmetic. +- Core gained dedicated unit tests for `CappedTextBuffer` and `ToolCallIdAllocator`, two small pure modules that previously had no direct coverage. diff --git a/changelog/0.2.0/2026-07-22-web-app.md b/changelog/0.2.0/2026-07-22-web-app.md new file mode 100644 index 0000000..19e053a --- /dev/null +++ b/changelog/0.2.0/2026-07-22-web-app.md @@ -0,0 +1,35 @@ +# Web App: workspace-grouped sidebar, key-aware model picker, chat rendering, letter avatars + +## Sidebar groups by Workspace + +The chat sidebar now groups conversations by Workspace by default, with a persisted toggle to switch back to Agent grouping. Workspace groups are labeled by the directory's basename (full path in the tooltip), ordered by their newest session; auto-created temp workspaces (`workspaces/tmp-<8hex>`) collapse into a single "Temp workspaces" group instead of one group per session. Each workspace header's "+" starts a new conversation pre-filled with that workspace; rows in workspace mode carry a small agent avatar so the agent context stays visible. Group collapse state persists per Project across reloads. Agent mode is unchanged. + +Groups gained polish and structure since: headers share one consistent hover height, the workspace folder icon opens and closes with the group, and both grouping modes support pinning groups to the top (persisted per Project). Sessions created by subagents and by scheduled tasks file into two collapsed folders — Subagents and Scheduled — parallel to Archived inside each group, and each group lists at most a page of sessions with a More row that loads further pages on demand (the session-list API gained `limit`/`offset` so the sidebar never fetches unbounded lists). + +## Model picker lists configured models first + +The chat model dropdown shows only models with a configured API key (a stored, masked key — an env-var name alone doesn't count) and a muted bottom row reveals the rest on demand, each marked with a key-with-slash icon (the localized label kept for assistive tech). The selected model and the project default are always visible, and when no model has a key yet the dropdown falls back to showing everything rather than rendering empty. + +## Letter avatars for custom model groups and Agents + +User-defined model provider groups all shared one generic icon, making same-named models across groups hard to tell apart in the model picker. Custom groups now render an initial-letter tile — the name's first grapheme on a softly tinted background derived from the name, with theme-aware colored ink that holds WCAG AA contrast in both light and dark mode — so each group keeps a stable, distinct look. Agent avatars adopt the same treatment, with the color keyed to the agent id so renames keep the color. + +## Collapsed rail navigation + +The collapsed sidebar is now a full navigation rail with eight entries in a fixed order — last conversation (the Project's newest non-archived session; disabled only when none exists, not while loading), new chat, Agents, Skills, Models, Costs, Traces, and Benchmark (previously missing) — each with bilingual hover tooltips, and a rail-level active highlight for any open conversation. + +## Cost center tooltips + +The daily-token chart's hover bubble follows the pointer at a lower-right offset, flipping near the edges so it never covers the hovered mark or clips; pointer moves position it imperatively (no re-render per mousemove). The Cache Read segment's bubble additionally shows the cache hit rate — cache reads over reads plus writes — computed and labeled identically to the Trace page's existing hit rate. + +## Settings forms + +In the model detail dialog, context window and max output tokens sit side by side with the Token unit inside each box and the explanation as placeholder plus hover tooltip; the model homepage moved to the dialog header as a button; the three price fields carry self-contained labels with no heading row; and vision support became an iPhone-style switch with no explanation text. Agent settings' runtime dropdowns no longer offer a default row — a reset link restores the inherited state instead — and all dropdowns render at the same text size as inputs. + +## Copied task stats are localized + +The per-task stats line produced by the copy-message action was hardcoded Chinese; it now goes through the dictionaries like every other UI string, so English sessions copy English stats. + +## Chat message rendering + +Links in chat messages always open in a new tab (`target="_blank" rel="noreferrer"`), covering explicit links and GFM autolinked bare URLs alike. The expanded subagent conversation now renders below the tool call's own arguments and output instead of above them, and every chat dropdown stays inside the viewport on mobile (three panels used to overflow by up to 143px, one of them side-scrolling the whole page). Long URLs and inline-code tokens wrap with `overflow-wrap: anywhere` — they break at the container edge without ever splitting ordinary Latin words inside CJK prose — and wide tables scroll inside the message instead of widening the page, matching the docs site's rule. The streaming pipeline's stable component maps are preserved, so code blocks keep their selection and copy state while a reply streams. diff --git a/changelog/0.2.0/README.md b/changelog/0.2.0/README.md new file mode 100644 index 0000000..d7a2d18 --- /dev/null +++ b/changelog/0.2.0/README.md @@ -0,0 +1,15 @@ +# Version 0.2.0 + +Unreleased. + +- [2026-07-22] Models and core: empty tool lists are omitted from LLM requests (fixing 400s from strict OpenAI-compatible servers), the default system prompt gains service-protection and API-key retry guardrails with the default port as a core SDK constant, a per-model max output tokens cap lands on the Models page, the thinking level moves to a conversation-time picker (low and above) that writes through to Agent settings, subagents inherit the parent session's model and thinking level, sessions record their origin in `session_meta.source` as the single source of truth, the SDK moves to AgentHub 0.4.1 and its supported-model registry drives a catalog refresh across every provider group (with the READMEs trimmed to the newest generation per vendor), and session-title generation folds into core's internal module. ([details](2026-07-22-models-and-core.md)) + +- [2026-07-22] Web App: the chat sidebar groups conversations by Workspace (with an Agent-mode toggle, group pinning, subagent/scheduled folders and paged loading), the collapsed sidebar becomes an eight-entry navigation rail with bilingual tooltips, the model picker lists key-configured models first, chat renders links in a new tab with clean CJK/URL wrapping and the subagent expansion below the tool's own output, mobile dropdowns stay inside the viewport, the Cost center's daily-token tooltip follows the pointer and shows the cache hit rate, the model and Agent settings forms were tightened, the copied task-stats line is localized, and custom model groups and Agents get initial-letter avatars. ([details](2026-07-22-web-app.md)) + +- [2026-07-22] Skills: vLLM and Ollama deployment plus LlamaFactory fine-tuning join the AI App Development group with a guided serving workflow that follows the user's engine preference, and `agenthub-models` tracks the AgentHub 0.4.1 API. ([details](2026-07-22-skills.md)) + +- [2026-07-22] Sites: the docs and landing navbars are now identical, the blog gains a Tech-practice category, pinned posts, author/date/copy-link metadata and a second AMD practice post, and the built-in Skills are listed in the READMEs and on the landing page. ([details](2026-07-22-sites-and-blog.md)) + +- [2026-07-22] Docs and examples: two new README roadmap items, and the self-improvement example reworked to genuinely evolve itself. ([details](2026-07-22-docs-and-examples.md)) + +- [2026-07-22] Tooling: server query-parameter validation hardening, and unit tests for two previously uncovered core modules. ([details](2026-07-22-tooling.md)) diff --git a/packages/landing/src/lib/strings-en.ts b/packages/landing/src/lib/strings-en.ts index 7eca0be..8b15c6f 100644 --- a/packages/landing/src/lib/strings-en.ts +++ b/packages/landing/src/lib/strings-en.ts @@ -342,7 +342,10 @@ export const en: Strings = { groups: [ { title: "Office Productivity", skills: ["data-analysis", "firecrawl"] }, { title: "Software Development", skills: ["web-design", "software-engineering"] }, - { title: "AI App Development", skills: ["penguin-sdk", "penguin-cli", "agenthub-models"] }, + { + title: "AI App Development", + skills: ["penguin-sdk", "penguin-cli", "agenthub-models", "vllm", "ollama", "llamafactory"], + }, { title: "Agent Tuning", skills: ["agent-creation", "benchmark-design", "agent-evaluation", "agent-optimization"], diff --git a/packages/landing/src/lib/strings.ts b/packages/landing/src/lib/strings.ts index a5512d5..fc77ecc 100644 --- a/packages/landing/src/lib/strings.ts +++ b/packages/landing/src/lib/strings.ts @@ -336,7 +336,10 @@ export const zh = { groups: [ { title: "办公效率", skills: ["data-analysis", "firecrawl"] }, { title: "软件开发", skills: ["web-design", "software-engineering"] }, - { title: "AI 应用开发", skills: ["penguin-sdk", "penguin-cli", "agenthub-models"] }, + { + title: "AI 应用开发", + skills: ["penguin-sdk", "penguin-cli", "agenthub-models", "vllm", "ollama", "llamafactory"], + }, { title: "Agent 调优", skills: ["agent-creation", "benchmark-design", "agent-evaluation", "agent-optimization"],