From 081fe2d172ce1e4950bd07eab1cd5ff0e7d1015b Mon Sep 17 00:00:00 2001 From: Jingzhe Xu <139672055+zzzbitz@users.noreply.github.com> Date: Thu, 30 Jul 2026 15:59:58 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E5=AE=8C=E5=96=84=E4=B8=A4=E9=98=B6?= =?UTF-8?q?=E6=AE=B5=20Agent=20=E8=87=AA=E8=BF=9B=E5=8C=96=20Pipeline?= =?UTF-8?q?=E3=80=81Skills=20=E4=B8=8E=20Benchmark=20(#129)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../2026-07-25-agent-tuning-pipeline.md | 15 + changelog/unreleased/README.md | 2 + packages/core/src/state/example-benchmark.ts | 111 +++--- packages/core/test/example-benchmark.test.ts | 33 +- packages/docs/content/self-improvement.en.md | 51 ++- packages/docs/content/self-improvement.zh.md | 51 ++- packages/docs/content/skills.en.md | 8 +- packages/docs/content/skills.zh.md | 8 +- packages/server/src/api/types.ts | 48 ++- packages/server/src/app.ts | 2 +- packages/server/src/http/routes/benchmarks.ts | 61 +++ .../server/src/services/benchmark-service.ts | 291 ++++++++++---- .../src/services/workspace-files-service.ts | 78 ++-- packages/server/test/benchmarks.test.ts | 209 +++++++--- packages/server/test/builtin-agents.test.ts | 1 + packages/server/test/workspace-files.test.ts | 5 + .../skills/skills/agent-creation/SKILL.md | 36 +- .../skills/skills/agent-evaluation/SKILL.md | 116 +++--- .../skills/skills/agent-optimization/SKILL.md | 200 +++++----- .../skills/skills/benchmark-design/SKILL.md | 227 ++++++----- packages/skills/test/skills.test.ts | 163 ++++++++ packages/web/src/api/endpoints.ts | 39 ++ .../features/benchmark/benchmark-metrics.ts | 49 +-- .../src/features/benchmark/benchmark-page.tsx | 310 ++++++++------- .../benchmark/benchmark-statement-browser.tsx | 367 ++++++++++++++++++ packages/web/src/features/chat/draft-view.tsx | 37 +- .../web/src/features/chat/example-tasks.ts | 17 + packages/web/src/lib/category-colors.ts | 4 +- packages/web/src/lib/format.ts | 4 +- packages/web/src/lib/strings-en.ts | 41 +- packages/web/src/lib/strings.ts | 53 ++- packages/web/test/benchmark-metrics.test.ts | 105 +++-- packages/web/test/example-tasks.test.ts | 105 +++++ packages/web/test/format.test.ts | 9 + 34 files changed, 2093 insertions(+), 763 deletions(-) create mode 100644 changelog/unreleased/2026-07-25-agent-tuning-pipeline.md create mode 100644 packages/web/src/features/benchmark/benchmark-statement-browser.tsx create mode 100644 packages/web/src/features/chat/example-tasks.ts create mode 100644 packages/web/test/example-tasks.test.ts diff --git a/changelog/unreleased/2026-07-25-agent-tuning-pipeline.md b/changelog/unreleased/2026-07-25-agent-tuning-pipeline.md new file mode 100644 index 0000000..37994c4 --- /dev/null +++ b/changelog/unreleased/2026-07-25-agent-tuning-pipeline.md @@ -0,0 +1,15 @@ +# Isolated Agent tuning pipeline + +The Web App now includes a runnable example that coordinates Agent creation, Benchmark construction, and score-driven Agent optimization without sharing private evaluation context between phases. + +## Web App + +The new draft-screen example launches each phase in an independent Penguin CLI Session derived from the active Project environment. Its Benchmark uses hidden context-to-action mappings so the optimizer can demonstrate measurable improvement from score-linked feedback. + +Benchmark construction now uses provisional Pilot evaluations to adjust one difficulty dimension at a time before freezing and recording the Formal Baseline. + +## Skills + +The Agent creation, Benchmark design, evaluation, and optimization Skills now define clearer ownership and access boundaries. Benchmark builders and optimizers delegate each Case run through an explicit Evaluator request protocol, while private Rubrics and Gold answers remain confined to the evaluation worker. + +Benchmark design now separates mutable Pilot calibration from the frozen Formal Baseline and prevents Rubric-only score reductions around observed answers. diff --git a/changelog/unreleased/README.md b/changelog/unreleased/README.md index ae0fdc7..eb88152 100644 --- a/changelog/unreleased/README.md +++ b/changelog/unreleased/README.md @@ -25,3 +25,5 @@ Changes since v0.1.4. The version number is assigned at release, when this folde - [2026-07-27] Windows: the `win32-x64` package bundles MinGit under `git/`, so `exec_command` has a real bash even on a machine with no Git for Windows — the shell stops depending on what happens to be installed. A user's own Git for Windows still wins (its MSYS userland is the fuller one); the bundle is the floor, and PowerShell is now reached only by npm installs. GPLv2 obligations are recorded in a new root `THIRD-PARTY-NOTICES.md`. ([details](2026-07-27-windows-bundled-shell.md)) - [2026-07-27] Sites: the 0.1.4 release post in both languages, with a capture script for its screenshots. ([details](2026-07-27-sites-and-blog.md)) + +- [2026-07-25] Agent tuning: the Web App gains an end-to-end example for creating, benchmarking, and optimizing an Agent through isolated CLI Sessions, while the built-in tuning Skills tighten phase ownership, private evaluation boundaries, and Evaluator dispatch contracts. ([details](2026-07-25-agent-tuning-pipeline.md)) diff --git a/packages/core/src/state/example-benchmark.ts b/packages/core/src/state/example-benchmark.ts index 2e53795..cd0f30d 100644 --- a/packages/core/src/state/example-benchmark.ts +++ b/packages/core/src/state/example-benchmark.ts @@ -8,10 +8,11 @@ * is a built-in example and the whole directory can be deleted or replaced. Only * default_agent gets this; ordinary Agents do not. * - * Scoring numbers are self-consistent: each case's score / cost / duration_ms is the - * **average** computed from its runs array, and each evaluation's totals are the sum over - * its cases (written this way so it already satisfies the scoreboard v2 convention, and - * tests can verify it). + * Scoring numbers follow the current Scoreboard contract: every Case is scored out of 100; + * Case metrics are model-written Run averages and Evaluation metrics are model-written Case + * averages. Cost ignores unknown values; Run cost preserves its recorded precision, Score + * averages keep two decimals, cost averages keep six, and durations are rounded to integer + * milliseconds. */ import fs from "node:fs/promises"; import path from "node:path"; @@ -42,11 +43,11 @@ Read the provided \`notes.txt\` in your workspace and write \`summary.md\` conta 2. A bullet list of the three most important facts. Keep the whole summary under 150 words. `, - rubric: `# Scoring rubric (max 5 points) + rubric: `# Scoring rubric (max 100 points) -- 2 pts: \`summary.md\` exists and stays under 150 words. -- 2 pts: The three bullet facts are accurate and taken from \`notes.txt\`. -- 1 pt: The overview paragraph is coherent and at most 3 sentences. +- 40 pts: \`summary.md\` exists and stays under 150 words. +- 40 pts: The three bullet facts are accurate and taken from \`notes.txt\`. +- 20 pts: The overview paragraph is coherent and at most 3 sentences. Award partial credit per item; the case score is the sum. `, }, @@ -60,20 +61,20 @@ Produce \`users_clean.csv\` where: 2. Exact duplicate rows are dropped, keeping the first occurrence. Do not change the column order. `, - rubric: `# Scoring rubric (max 5 points) + rubric: `# Scoring rubric (max 100 points) -- 2 pts: \`users_clean.csv\` exists and keeps the original column order. -- 2 pts: Emails are lowercased, empty-email rows removed, duplicates dropped (first kept). -- 1 pt: No unrelated rows or columns were modified. +- 40 pts: \`users_clean.csv\` exists and keeps the original column order. +- 40 pts: Emails are lowercased, empty-email rows removed, duplicates dropped (first kept). +- 20 pts: No unrelated rows or columns were modified. Award partial credit per item; the case score is the sum. `, }, ]; -/** Raw result of a single run (a runs element in scoreboard v2). */ +/** Raw result of a single Run in the current Scoreboard format. */ interface ExampleRun { score: number; - cost: number; + cost: number | null; duration_ms: number; session_id: string; } @@ -81,14 +82,14 @@ interface ExampleRun { /** * Raw runs for the three sample evaluations (case-level and evaluation-level metrics are * computed from these, keeping the numbers self-consistent). Each carries the model actually - * used for that round (paired, since the evaluation center's chart splits series by model); - * the examples all use deepseek-v4-pro (a single model, single series). + * used for that round; the examples all use deepseek-v4-pro at medium thinking. */ const EXAMPLE_EVALUATIONS: Array<{ time: string; version: number; provider: string; model_id: string; + thinking_level: string; summary_title: string; summary: string; cases: Array<{ case: string; runs: ExampleRun[] }>; @@ -98,6 +99,7 @@ const EXAMPLE_EVALUATIONS: Array<{ version: 1, provider: "deepseek", model_id: "deepseek-v4-pro", + thinking_level: "medium", summary_title: "Baseline before any optimization", summary: "Example data (not a real evaluation): baseline scores of the built-in sample " + @@ -108,13 +110,13 @@ const EXAMPLE_EVALUATIONS: Array<{ case: "CASE-001-file-summary", runs: [ { - score: 2.5, + score: 50, cost: 0.012, duration_ms: 42000, session_id: "session-2026-07-14-09-05-11-1a2b3c01", }, { - score: 3.5, + score: 70, cost: 0.014, duration_ms: 48000, session_id: "session-2026-07-14-09-13-27-1a2b3c02", @@ -125,14 +127,14 @@ const EXAMPLE_EVALUATIONS: Array<{ case: "CASE-002-data-cleanup", runs: [ { - score: 3.0, - cost: 0.018, + score: 60, + cost: null, duration_ms: 66000, session_id: "session-2026-07-14-09-21-45-1a2b3c03", }, { - score: 3.0, - cost: 0.022, + score: 60, + cost: null, duration_ms: 74000, session_id: "session-2026-07-14-09-28-52-1a2b3c04", }, @@ -145,6 +147,7 @@ const EXAMPLE_EVALUATIONS: Array<{ version: 2, provider: "deepseek", model_id: "deepseek-v4-pro", + thinking_level: "medium", summary_title: "Added an explicit planning step", summary: "Example data (not a real evaluation): after adding an explicit planning step to the " + @@ -155,13 +158,13 @@ const EXAMPLE_EVALUATIONS: Array<{ case: "CASE-001-file-summary", runs: [ { - score: 3.5, + score: 70, cost: 0.011, duration_ms: 39000, session_id: "session-2026-07-15-09-04-33-2b3c4d01", }, { - score: 4.0, + score: 80, cost: 0.013, duration_ms: 45000, session_id: "session-2026-07-15-09-12-08-2b3c4d02", @@ -172,13 +175,13 @@ const EXAMPLE_EVALUATIONS: Array<{ case: "CASE-002-data-cleanup", runs: [ { - score: 3.5, + score: 70, cost: 0.016, duration_ms: 60000, session_id: "session-2026-07-15-09-19-40-2b3c4d03", }, { - score: 4.0, + score: 80, cost: 0.02, duration_ms: 68000, session_id: "session-2026-07-15-09-26-59-2b3c4d04", @@ -192,6 +195,7 @@ const EXAMPLE_EVALUATIONS: Array<{ version: 3, provider: "deepseek", model_id: "deepseek-v4-pro", + thinking_level: "medium", summary_title: "Verify deliverables before finishing", summary: "Example data (not a real evaluation): after instructing the agent to verify its " + @@ -203,13 +207,13 @@ const EXAMPLE_EVALUATIONS: Array<{ case: "CASE-001-file-summary", runs: [ { - score: 4.0, + score: 80, cost: 0.01, duration_ms: 36000, session_id: "session-2026-07-16-09-03-21-3c4d5e01", }, { - score: 4.5, + score: 90, cost: 0.012, duration_ms: 40000, session_id: "session-2026-07-16-09-10-46-3c4d5e02", @@ -220,13 +224,13 @@ const EXAMPLE_EVALUATIONS: Array<{ case: "CASE-002-data-cleanup", runs: [ { - score: 4.5, + score: 90, cost: 0.015, duration_ms: 55000, session_id: "session-2026-07-16-09-18-02-3c4d5e03", }, { - score: 4.0, + score: 80, cost: 0.017, duration_ms: 61000, session_id: "session-2026-07-16-09-25-30-3c4d5e04", @@ -237,24 +241,31 @@ const EXAMPLE_EVALUATIONS: Array<{ }, ]; -/** Round floats to 1e-6 (so binary error from averaging/summing isn't persisted to disk). */ -function round(v: number): number { - return Math.round(v * 1e6) / 1e6; +function roundTwo(v: number): number { + return Math.round(v * 100) / 100; } -function average(values: number[]): number { - return round(values.reduce((a, b) => a + b, 0) / values.length); +function averageTwo(values: number[]): number { + return roundTwo(values.reduce((a, b) => a + b, 0) / values.length); } -function sum(values: number[]): number { - return round(values.reduce((a, b) => a + b, 0)); +function averageSix(values: number[]): number { + return Math.round((values.reduce((a, b) => a + b, 0) / values.length) * 1_000_000) / 1_000_000; +} + +function averageDuration(values: number[]): number { + return Math.round(values.reduce((a, b) => a + b, 0) / values.length); +} + +function averageKnownCost(values: Array): number | null { + const known = values.filter((value): value is number => value !== null); + return known.length > 0 ? averageSix(known) : null; } /** - * Builds the scoreboard object from raw runs data: each case's three metrics are the - * average of its runs, and each evaluation's metrics are the sum of its cases' averages - * (following the scoreboard v2 convention). Exported so tests can verify the numbers - * are self-consistent. + * Builds the example Scoreboard exactly as the model is instructed to write it. This helper + * exists only to provision deterministic sample data; runtime readers trust the stored values + * and never call it to recompute a user Scoreboard. */ export function buildExampleScoreboard(): { evaluations: Array<{ @@ -262,15 +273,16 @@ export function buildExampleScoreboard(): { version: number; provider: string; model_id: string; + thinking_level: string; summary_title: string; summary: string; score: number; - cost: number; + cost: number | null; duration_ms: number; cases: Array<{ case: string; score: number; - cost: number; + cost: number | null; duration_ms: number; runs: ExampleRun[]; }>; @@ -280,9 +292,9 @@ export function buildExampleScoreboard(): { evaluations: EXAMPLE_EVALUATIONS.map((e) => { const cases = e.cases.map((c) => ({ case: c.case, - score: average(c.runs.map((r) => r.score)), - cost: average(c.runs.map((r) => r.cost)), - duration_ms: average(c.runs.map((r) => r.duration_ms)), + score: averageTwo(c.runs.map((r) => r.score)), + cost: averageKnownCost(c.runs.map((r) => r.cost)), + duration_ms: averageDuration(c.runs.map((r) => r.duration_ms)), runs: c.runs, })); return { @@ -290,11 +302,12 @@ export function buildExampleScoreboard(): { version: e.version, provider: e.provider, model_id: e.model_id, + thinking_level: e.thinking_level, summary_title: e.summary_title, summary: e.summary, - score: sum(cases.map((c) => c.score)), - cost: sum(cases.map((c) => c.cost)), - duration_ms: sum(cases.map((c) => c.duration_ms)), + score: averageTwo(cases.map((c) => c.score)), + cost: averageKnownCost(cases.map((c) => c.cost)), + duration_ms: averageDuration(cases.map((c) => c.duration_ms)), cases, }; }), diff --git a/packages/core/test/example-benchmark.test.ts b/packages/core/test/example-benchmark.test.ts index 168154b..9ec8802 100644 --- a/packages/core/test/example-benchmark.test.ts +++ b/packages/core/test/example-benchmark.test.ts @@ -46,7 +46,7 @@ async function exists(p: string): Promise { interface RunScore { score: number; - cost: number; + cost: number | null; duration_ms: number; session_id: string; } @@ -59,6 +59,7 @@ interface Evaluation extends Omit { version: number; provider: string; model_id: string; + thinking_level: string; summary_title: string; summary: string; cases: CaseScore[]; @@ -88,6 +89,7 @@ describe("example benchmark provisioning", () => { const rubric = await fs.readFile(path.join(dir, caseId, "rubric", "README.md"), "utf8"); expect(statement.length).toBeGreaterThan(50); expect(rubric).toContain("pts"); + expect(rubric).toContain("max 100 points"); } // scoreboard.yaml: 3 evaluations, with version/time increasing and scores rising, and @@ -96,6 +98,7 @@ describe("example benchmark provisioning", () => { evaluations: Evaluation[]; }; expect(scoreboard.evaluations).toHaveLength(3); + expect(JSON.stringify(scoreboard)).not.toContain("max_score"); expect(scoreboard.evaluations.map((e) => e.version)).toEqual([1, 2, 3]); const times = scoreboard.evaluations.map((e) => new Date(e.time).getTime()); expect(times[0]!).toBeLessThan(times[1]!); @@ -112,6 +115,7 @@ describe("example benchmark provisioning", () => { ]); for (const e of scoreboard.evaluations) { expect(e.provider).toBe("deepseek"); + expect(e.thinking_level).toBe("medium"); expect(e.summary_title.length).toBeGreaterThan(0); expect(e.summary.toLowerCase()).toContain("example"); expect(e.cases).toHaveLength(2); @@ -124,19 +128,30 @@ describe("example benchmark provisioning", () => { } }); - it("scoreboard numbers are self-consistent (case = avg of runs, evaluation = sum of cases)", async () => { + it("scoreboard numbers preserve Run cost precision and use model-written averages", async () => { const { evaluations } = buildExampleScoreboard(); + expect(evaluations[0]?.cases[0]?.runs.map((run) => run.cost)).toEqual([0.012, 0.014]); const avg = (vals: number[]): number => vals.reduce((a, b) => a + b, 0) / vals.length; - const sum = (vals: number[]): number => vals.reduce((a, b) => a + b, 0); + const knownCostAvg = (vals: Array): number | null => { + const known = vals.filter((value): value is number => value !== null); + return known.length > 0 ? avg(known) : null; + }; for (const e of evaluations) { for (const c of e.cases) { - expect(c.score).toBeCloseTo(avg(c.runs.map((r) => r.score)), 6); - expect(c.cost).toBeCloseTo(avg(c.runs.map((r) => r.cost)), 6); - expect(c.duration_ms).toBeCloseTo(avg(c.runs.map((r) => r.duration_ms)), 6); + expect(c.runs.every((run) => run.score >= 0 && run.score <= 100)).toBe(true); + expect(c.score).toBe(Math.round(avg(c.runs.map((r) => r.score)) * 100) / 100); + const expectedCost = knownCostAvg(c.runs.map((r) => r.cost)); + expect(c.cost).toBe( + expectedCost === null ? null : Math.round(expectedCost * 1_000_000) / 1_000_000, + ); + expect(c.duration_ms).toBe(Math.round(avg(c.runs.map((r) => r.duration_ms)))); } - expect(e.score).toBeCloseTo(sum(e.cases.map((c) => c.score)), 6); - expect(e.cost).toBeCloseTo(sum(e.cases.map((c) => c.cost)), 6); - expect(e.duration_ms).toBeCloseTo(sum(e.cases.map((c) => c.duration_ms)), 6); + expect(e.score).toBe(Math.round(avg(e.cases.map((c) => c.score)) * 100) / 100); + const expectedCost = knownCostAvg(e.cases.map((c) => c.cost)); + expect(e.cost).toBe( + expectedCost === null ? null : Math.round(expectedCost * 1_000_000) / 1_000_000, + ); + expect(e.duration_ms).toBe(Math.round(avg(e.cases.map((c) => c.duration_ms)))); } }); diff --git a/packages/docs/content/self-improvement.en.md b/packages/docs/content/self-improvement.en.md index 2c7095a..8ff8cb6 100644 --- a/packages/docs/content/self-improvement.en.md +++ b/packages/docs/content/self-improvement.en.md @@ -3,27 +3,40 @@ title: Self-Improvement description: The Skill-orchestrated Benchmark and optimization loop: score, improve, snapshot, roll back. --- -Self-improvement in PenguinHarness is not carried by special-purpose engine code — it is carried by Skills orchestrating the ordinary Agent machinery: evaluations are ordinary Sessions, optimization is ordinary file editing, and orchestration uses the built-in `run_subagent` tool. The direct payoff is that the whole process shares the same observability and recovery machinery as everyday runs. +Self-improvement in PenguinHarness uses Skills to orchestrate the ordinary Agent machinery: evaluations are ordinary Sessions and optimization is ordinary file editing. Evaluation construction and optimization run in two independent top-level Sessions, while individual evaluations are delegated through the built-in `run_subagent` tool. Top-level prompts provide the Agent, Benchmark, capability, score, and round settings for the task; Skills own call relationships, calibration, Freeze, protocol, repair, rollback, and reporting. -## Three roles +## Roles and call relationships | Role | Responsibility | | --- | --- | +| Builder | Top-level Agent that directly follows `agent-creation` and then `benchmark-design` | | Target Agent | The Agent being improved; runs evaluation tasks only inside its own Workspace | -| Evaluator | Runs and scores one Benchmark Case run | -| Optimizer | Drives the whole optimization loop | +| Evaluator | Leaf worker created through `run_subagent`; runs and scores one Benchmark Case run | +| Optimizer | New top-level Agent that directly follows `agent-optimization` | -The roles are defined by Skills, not hardcoded: the Evaluator follows the `agent-evaluation` Skill, the Optimizer follows the `agent-optimization` Skill. This applies the design principle stated in the [Configuration Reference](/configuration) — an Agent's behavior is editable files on disk, which is what makes Agents improvable by Agents. +The Builder and Optimizer directly follow their Skills in their own top-level Sessions. Evaluators are created through `run_subagent`; each follows `agent-evaluation` and uses the Penguin CLI to launch the specified Target Agent in an isolated Workspace identified by an absolute path. The Penguin CLI launches the Target Agent for the requested Case run. -## The loop +## Two independent steps -1. `benchmark-design` builds a multi-Case capability Benchmark: repeated independent runs, with a traceable baseline calibrated first; -2. The Optimizer orchestrates Evaluators in parallel via the `run_subagent` tool, covering the Case × runs matrix; -3. Scores plus their linked Traces show where points were lost; -4. The Optimizer edits the Target Agent's editable state — `AGENTS.md`, Skills, config — to produce version N+1; -5. A Snapshot is taken before each round; the candidate version is kept only if the total score strictly improves, otherwise rolled back. +The first top-level Session creates the Agent and its capability evaluation. The Builder first uses `agent-creation`, then uses `benchmark-design` to build a multi-Case Benchmark. It may build the complete initial Case set before Pilot 1 and may refine multiple Cases or difficulty dimensions in a later iteration. The evaluation contract and private standard must be clear and fixed, while the public Statement need not uniquely determine the Gold. A Benchmark may use incomplete public information, conflicting signals, and a fixed private decision standard when that standard expresses a reusable policy, priority, or inference boundary and is not rewritten after seeing the run's answer. -Benchmark optimization mode requires a complete baseline series in the scoreboard — without a calibrated baseline there is no improvement to compare against. Besides this loop, `agent-optimization` also supports a one-shot feedback mode: a concrete correction is applied directly as edits to the Target Agent's state, without going through the evaluation loop. +Before the first dispatch of every new or changed Case, the Builder checks that the Statement is internally coherent, the Rubric agrees with the current Statement and fixed private standard, and every scoring item relies only on defined, provided, or explicitly private premises; this does not require the public materials to reproduce the private standard. It repeats the full review across all Cases before Freeze. Most points should rest on decisions or concise artifacts for which the intended behavior and a plausible shortcut produce different results, rather than giving a high floor for format, evidence enumeration, or analysis completeness. + +Before each calibration dispatch, the Builder predicts the result produced by the observed Trace strategy, the different result produced by the desired behavior, and the score range affected. Adding another public rule, exception, source, or check that the model can directly execute does not automatically increase difficulty. If both strategies still reach the same scored result, the Builder chooses another refinement. + +The Pilot score is a desired target: meeting it permits an early Freeze; otherwise the Builder completes the configured number of valid Pilot iterations and freezes the lowest-scoring valid revision. The Builder temporarily retains only the current lowest valid revision, then removes that copy and other calibration scaffolding after recording the Formal Baseline. Freeze is followed by a fresh complete Formal matrix. Every valid Formal Baseline is recorded even when its score misses the desired target. + +After the user confirms that step is complete, they start the second top-level Session in a new conversation. The Optimizer checks the Benchmark and its first complete Formal Baseline before following `agent-optimization`: + +1. orchestrate Evaluators in parallel through `run_subagent`, covering the Case × runs matrix; +2. use scores and linked Traces to propose one bounded Candidate; +3. edit the Target Agent's editable state — `AGENTS.md`, Skills, config — to produce version N+1; +4. keep the Candidate only when its Evaluation score strictly improves; otherwise roll it back; +5. stop early when the desired score is reached, or complete the configured number of valid Candidate rounds and retain the highest-scoring Reference. + +Invalid evaluations and correction reruns do not count toward the round limit. On an execution failure, the Optimizer keeps the same Candidate and repairs only the missing cell; it keeps trying while each attempt follows a new diagnosis and applies a distinct safe repair. Both Builder and Optimizer validate that the complete Evaluator response is plain protocol YAML before reading status or score; if formatting is invalid, that same Evaluator resends from its existing result without rerunning the Target Agent. + +Every accepted Candidate is appended to and verified in the Scoreboard immediately. A strictly higher Evaluation score decides acceptance; whether the predicted Case behavior changed is reported separately so unrelated single-run variation is not presented as causal evidence. Agent optimization requires a complete Formal Baseline in the Scoreboard — without one there is no improvement to compare against. ## Benchmark storage @@ -35,18 +48,20 @@ benchmarks// ├── / │ ├── statement/ # the task given to the Target Agent │ └── rubric/ # private scoring rubric, isolated from the Target Agent -└── scoreboard.yaml # evaluation records (v2 format) +└── scoreboard.yaml # evaluation records (current format) ``` The separation of `rubric/` from `statement/` is deliberate: the Target Agent sees only the task statement and never touches the scoring rubric. -Each evaluation record in `scoreboard.yaml` (v2 format) is timestamped and carries: +Each evaluation record in `scoreboard.yaml` is timestamped and carries: -- the paired model reference `(provider, model_id)` used for the round; +- the evaluation runtime: a user-specified `(provider, model_id)` pair takes priority, otherwise the pair is inherited from the Builder Session; `thinking_level` is read from the Target Agent config and does not depend on Trace metadata; - `summary_title` and `summary` (the round's conclusion and the hypothesis for the next one); -- total score, cost, and duration — Case-level metrics are the average over its runs, evaluation-level metrics are the sum over its Cases; +- Score, cost, and duration averages written by the model — Case-level values average Runs and Evaluation-level values average Cases; Run cost preserves its recorded precision, cost averages ignore `null` inputs and remain `null` only when every contributing cost is unknown; Score uses two decimals, cost averages use six decimals, and `duration_ms` is an integer; - per-Case run details, each run recording `score`, `cost`, `duration_ms`, and `session_id`. +Every Run and every Case has a fixed maximum Score of 100, so Scoreboard entries do not carry `max_score`. The server and Web UI trust the stored aggregate values and do not recompute or cross-check them. Old Scoreboard formats are not migrated or backfilled. + The built-in `default_agent` ships with an example Benchmark (`packages/core/src/state/example-benchmark.ts`) so the evaluation pages have data out of the box; the whole directory can be deleted or replaced at any time. ## Snapshots and versions @@ -57,7 +72,7 @@ Before each optimization round, the Agent State is packed into `snapshots/v/ ├── / │ ├── statement/ # 交给 Target Agent 的任务描述 │ └── rubric/ # 私有评分标准,对 Target Agent 隔离 -└── scoreboard.yaml # 评测记录(v2 格式) +└── scoreboard.yaml # 当前格式的评测记录 ``` `rubric/` 与 `statement/` 的隔离是刻意设计:Target Agent 只能看到题面,永远接触不到评分标准。 -`scoreboard.yaml`(v2 格式)中的每条评测记录带时间戳,并记录: +`scoreboard.yaml` 中的每条评测记录带时间戳,并记录: -- 本轮使用的模型成对引用 `(provider, model_id)`; +- 本轮 Runtime:用户显式指定的 `(provider, model_id)` 成对值优先,否则继承 Builder Session;`thinking_level` 从 Target Agent 配置读取,不依赖 Trace 元数据; - `summary_title` 与 `summary`(本轮结论与下一轮假设); -- 总分、成本与耗时——Case 级指标是各次运行的平均值,评测级指标是各 Case 的加和; +- 由模型写入的 Score、成本与耗时平均值——Case 级对 Runs 求平均,Evaluation 级对 Cases 求平均;单次 Run 成本保留记录中的原始精度,成本平均值忽略 `null`,全部未知时才为 `null`;Score 保留两位小数,成本平均值保留六位小数,`duration_ms` 取整; - 每个 Case 的逐次运行明细,每次运行含 `score`、`cost`、`duration_ms` 与 `session_id`。 +每个 Run 和每个 Case 都固定满分 100,因此 Scoreboard 不再记录 `max_score`。服务端与 Web UI 直接信任已写入的聚合值,不重算、不交叉校验;旧 Scoreboard 不迁移、不回填。 + 内置的 `default_agent` 预置了一个示例 Benchmark(`packages/core/src/state/example-benchmark.ts`),评测页面开箱即有数据;整个目录可随时删除或替换。 ## Snapshot 与版本 @@ -57,7 +72,7 @@ benchmarks// - 每次 Evaluator 运行都是一个普通的 Session,留有完整 Trace; - scoreboard 记录通过 `session_id` 链接回这些 Session,见 [Session 与 Trace](/sessions-and-traces); -- Web 的评测页面是这些文件的只读视图,见 [Web App 指南](/web-app)。 +- Web 的评测页面是这些文件的只读视图;折线图只展示 Score,明细表将模型 ID 与推理强度分列显示。见 [Web App 指南](/web-app)。 分数不是黑盒输出:任何一个数字都可以回溯到产生它的那次运行。 @@ -68,6 +83,6 @@ benchmarks// | `agent-creation` | 把需求变成可用的 Agent:撰写其 `AGENTS.md`、安装所需 Skill | | `benchmark-design` | 设计并校准多 Case 的能力 Benchmark | | `agent-evaluation` | 隔离执行并评分一次 Benchmark Case 运行 | -| `agent-optimization` | 根据反馈或 Benchmark 结果改进 Agent | +| `agent-optimization` | 根据 Benchmark 结果改进 Agent | Skill 的组织与安装方式见[技能系统](/skills)。 diff --git a/packages/docs/content/skills.en.md b/packages/docs/content/skills.en.md index 40172da..33067ff 100644 --- a/packages/docs/content/skills.en.md +++ b/packages/docs/content/skills.en.md @@ -69,10 +69,10 @@ The built-in Skills, by group (the group manifest is `SKILL_GROUPS` in `packages | | `vllm` | Deploy and serve LLMs with vLLM behind an OpenAI-compatible endpoint, with tool calling enabled for agent workloads | | | `ollama` | Deploy and serve local models with Ollama: pull and run them, then expose the OpenAI-compatible endpoint to apps and agents | | | `llamafactory` | Fine-tune LLMs with LlamaFactory: register datasets, train via YAML configs, merge LoRA adapters and serve the result | -| Agent Tuning | `agent-creation` | Turn a user requirement into a concrete agent: write the target agent's AGENTS.md and install the skills it needs | -| | `benchmark-design` | Design and calibrate a multi-Case capability Benchmark with repeated independent evaluations and a traceable baseline | -| | `agent-evaluation` | Run and score exactly one Benchmark Case run, with CLI execution, Trace provenance checks and private Rubric isolation | -| | `agent-optimization` | Improve an Agent State from direct feedback or versioned multi-Case Benchmark scores and score-linked Traces | +| Agent Tuning | `agent-creation` | Create or configure an Agent State from a user requirement by writing AGENTS.md, setting identity metadata and installing needed Skills | +| | `benchmark-design` | Design and calibrate a multi-Case capability Benchmark for a specified Agent and establish a traceable Formal Baseline | +| | `agent-evaluation` | Internal leaf worker that executes and privately scores exactly one Case run from a complete evaluation protocol | +| | `agent-optimization` | Improve a specified Agent from a complete current baseline on a frozen Benchmark | ## Writing and optimizing Skills diff --git a/packages/docs/content/skills.zh.md b/packages/docs/content/skills.zh.md index 25e5d68..e54320a 100644 --- a/packages/docs/content/skills.zh.md +++ b/packages/docs/content/skills.zh.md @@ -69,10 +69,10 @@ Skill 库以 npm 包 `@prismshadow/penguin-skills` 发布,tarball 直接携带 | | `vllm` | 用 vLLM 部署与服务 LLM,提供 OpenAI 兼容端点,并为 Agent 负载启用工具调用 | | | `ollama` | 用 Ollama 部署与运行本地模型,把 OpenAI 兼容端点接入应用与 Agent | | | `llamafactory` | 用 LlamaFactory 微调 LLM:注册数据集、以 YAML 配置训练、合并 LoRA 适配器并部署产物 | -| Agent 调优 | `agent-creation` | 把用户需求变成具体的 Agent:撰写目标 Agent 的 AGENTS.md 并安装所需 Skill | -| | `benchmark-design` | 设计并校准多 Case 的能力评测 Benchmark,含重复独立评测与可追溯基线 | -| | `agent-evaluation` | 隔离执行并评分单个 Benchmark Case:CLI 执行、Trace 溯源检查、Rubric 私有隔离 | -| | `agent-optimization` | 依据直接反馈或带版本的多 Case Benchmark 分数与关联 Trace 改进 Agent State | +| Agent 调优 | `agent-creation` | 根据需求创建或配置 Agent State,编写 AGENTS.md、设置身份信息并安装所需 Skill | +| | `benchmark-design` | 为指定 Agent 设计并校准多 Case Benchmark,建立可追溯的 Formal Baseline | +| | `agent-evaluation` | 内部叶子执行器:根据完整评测协议隔离执行并私密评分一个 Case Run | +| | `agent-optimization` | 基于冻结 Benchmark 的完整当前基线改进指定 Agent | ## 编写与优化 diff --git a/packages/server/src/api/types.ts b/packages/server/src/api/types.ts index 3f184eb..dbc7cc1 100644 --- a/packages/server/src/api/types.ts +++ b/packages/server/src/api/types.ts @@ -1214,22 +1214,23 @@ export interface AgentImportResponse { /** Raw result of a single run (a scoreboard per-case runs[] entry). */ export interface BenchmarkRunScore { score: number; - cost?: number; - durationMs?: number; + /** Run cost, or null when unavailable. */ + cost: number | null; + durationMs: number; /** Id of the Session under test in this run (links to Trace). */ - sessionId?: string; + sessionId: string; } export interface BenchmarkCaseScore { case: string; - /** Per-case score = average of runs (equals that single run's score under the legacy single-run format). */ + /** Model-written average of this Case's Run scores, on the fixed 0..100 scale. */ score: number; - cost?: number; - durationMs?: number; - /** For legacy format compatibility: per-case single Session id (new format keeps it inside runs[]). */ - sessionId?: string; - /** Raw results per run; unset under the legacy format (the server backfills one entry when parsing as a single run). */ - runs?: BenchmarkRunScore[]; + /** Model-written average of known Run costs; null when every Run cost is unknown. */ + cost: number | null; + /** Model-written average of Run durations, rounded to an integer. */ + durationMs: number; + /** Raw results per Run. */ + runs: BenchmarkRunScore[]; } export interface BenchmarkEvaluation { @@ -1240,15 +1241,19 @@ export interface BenchmarkEvaluation { /** Evaluation summary body: how the score was derived, what optimizations were made to the Agent this round (required when generating, tolerated as unset when displaying). */ summary?: string; /** Model actually used for this evaluation round (upstream id, paired with provider; the chart series is split by model). */ - modelId?: string; + modelId: string; /** Provider group for `modelId`. */ - provider?: string; + provider: string; + /** Thinking level read from the unchanged Target Agent configuration. */ + thinkingLevel: string; /** Agent State version number under test. */ - version?: number; - /** Total score (sum of per-case scores; max score defined by the scoring rubric). */ + version: number; + /** Model-written average of Case scores, on the fixed 0..100 scale. */ score: number; - cost?: number; - durationMs?: number; + /** Model-written average of known Case costs; null when every Case cost is unknown. */ + cost: number | null; + /** Model-written average of Case durations, rounded to an integer. */ + durationMs: number; cases: BenchmarkCaseScore[]; } @@ -1270,6 +1275,17 @@ export interface BenchmarksResponse { benchmarks: BenchmarkSummary[]; } +/** Public Benchmark Case metadata. Rubric and Gold content are never included. */ +export interface BenchmarkCaseSummary { + id: string; + /** First Markdown heading with an optional leading "Case N:" removed; falls back to id. */ + title: string; +} + +export interface BenchmarkCasesResponse { + cases: BenchmarkCaseSummary[]; +} + // --------------------------------------------------------------------------- // Skill library and Agent's installed Skills // --------------------------------------------------------------------------- diff --git a/packages/server/src/app.ts b/packages/server/src/app.ts index 72a6ca4..1745886 100644 --- a/packages/server/src/app.ts +++ b/packages/server/src/app.ts @@ -154,7 +154,7 @@ export function buildAppDeps(config: ServerConfig, overrides: BuildDepsOverrides // Per-process secret: preview tokens are short-lived, so losing them on restart is // harmless and there is nothing to persist or rotate. const previewTokens = createPreviewTokenSigner(); - const benchmarks = new BenchmarkService(config.root); + const benchmarks = new BenchmarkService(config.root, workspaceFiles); const snapshots = new SnapshotService(config.root); const usageService = new UsageService( usageRepo, diff --git a/packages/server/src/http/routes/benchmarks.ts b/packages/server/src/http/routes/benchmarks.ts index f66a12d..51f1f84 100644 --- a/packages/server/src/http/routes/benchmarks.ts +++ b/packages/server/src/http/routes/benchmarks.ts @@ -1,6 +1,9 @@ /** * Benchmark scoring routes: * GET /api/projects/:p/agents/:a/benchmarks (any member, read-only) + * GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases + * GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/files + * GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/files/content * Returns the Agent's Benchmark list (title/description from benchmark_config.toml) * along with the evaluations[] from scoreboard.yaml. */ @@ -9,6 +12,8 @@ import type { AppEnv } from "../../auth/middleware.js"; import type { AppDeps } from "../../app.js"; import { requireValidId } from "../validate.js"; +const TEXT_PREVIEW_BYTES = 256 * 1024; + export function benchmarksRoutes(deps: AppDeps): Hono { const app = new Hono(); @@ -20,5 +25,61 @@ export function benchmarksRoutes(deps: AppDeps): Hono { return c.json(await deps.benchmarks.list(projectId, agentId)); }); + app.get("/:benchmarkId/cases", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + const benchmarkId = requireValidId(c, "benchmarkId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + return c.json(await deps.benchmarks.listCases(projectId, agentId, benchmarkId)); + }); + + app.get("/:benchmarkId/cases/:caseId/files", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + const benchmarkId = requireValidId(c, "benchmarkId"); + const caseId = requireValidId(c, "caseId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + return c.json( + await deps.benchmarks.listCaseFiles( + projectId, + agentId, + benchmarkId, + caseId, + c.req.query("path") ?? "", + ), + ); + }); + + app.get("/:benchmarkId/cases/:caseId/files/content", async (c) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + const benchmarkId = requireValidId(c, "benchmarkId"); + const caseId = requireValidId(c, "caseId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + const download = c.req.query("download") === "1"; + const boundedPreview = !download && c.req.query("preview") === "1"; + const { data, fileName, contentType, scriptable, truncated } = + await deps.benchmarks.readCaseFile( + projectId, + agentId, + benchmarkId, + caseId, + c.req.query("path") ?? "", + boundedPreview ? { maxBytes: TEXT_PREVIEW_BYTES } : undefined, + ); + return new Response(new Uint8Array(data), { + status: 200, + headers: { + "Content-Type": !download && scriptable ? "text/plain; charset=utf-8" : contentType, + "Content-Disposition": `${download ? "attachment" : "inline"}; filename*=UTF-8''${encodeURIComponent(fileName)}`, + "X-Content-Type-Options": "nosniff", + ...(truncated ? { "X-Content-Truncated": "1" } : {}), + }, + }); + }); + return app; } diff --git a/packages/server/src/services/benchmark-service.ts b/packages/server/src/services/benchmark-service.ts index a19e55e..84bdb8e 100644 --- a/packages/server/src/services/benchmark-service.ts +++ b/packages/server/src/services/benchmark-service.ts @@ -1,16 +1,15 @@ /** * Benchmark score reading (read-only display): walks `benchmarks//`, reads - * `benchmark_config.toml` (title, - * description, evaluation Model, per-case run count `runs`) and `scoreboard.yaml` - * (evaluations[], scoreboard v2: each case carries a runs array and a summary). + * `benchmark_config.toml` (title, description, per-case run count `runs`) and + * `scoreboard.yaml` (evaluations[], each case carries its model-written averages + * and a runs array). * Content is created and refined by benchmark_builder; the server only reads it. * Missing or corrupt files always degrade gracefully (title falls back to the * directory name, scores come back empty) rather than throwing. * - * The three per-case metrics trust the file's own values; when missing they're - * computed as the average over the runs array. The old format (no runs at the - * case level, a single session_id) is parsed as a single run — the server backfills - * one run entry. + * Case and Evaluation averages are authoritative file values. The server validates + * the current shape but never recomputes aggregates and does not migrate or backfill + * old Scoreboard formats. * Docs: /docs/self-improvement § "Benchmark storage". */ import fs from "node:fs/promises"; @@ -20,11 +19,22 @@ import { parse as parseYaml } from "yaml"; import { benchmarksDir } from "@prismshadow/penguin-core"; import type { BenchmarkCaseScore, + BenchmarkCaseSummary, + BenchmarkCasesResponse, BenchmarkEvaluation, BenchmarkRunScore, BenchmarkSummary, BenchmarksResponse, + WorkspaceFilesResponse, } from "../api/types.js"; +import type { + WorkspaceFileContent, + WorkspaceFileReadOptions, + WorkspaceFilesService, +} from "./workspace-files-service.js"; +import { HttpError } from "../http/errors.js"; + +const STATEMENT_TITLE_READ_BYTES = 64 * 1024; function asRecord(v: unknown): Record { return v !== null && typeof v === "object" && !Array.isArray(v) @@ -36,110 +46,151 @@ function numberOr(v: unknown): number | undefined { return typeof v === "number" && Number.isFinite(v) ? v : undefined; } +function scoreOr(v: unknown): number | undefined { + const value = numberOr(v); + return value !== undefined && value >= 0 && value <= 100 ? value : undefined; +} + +function nonNegativeOr(v: unknown): number | undefined { + const value = numberOr(v); + return value !== undefined && value >= 0 ? value : undefined; +} + +function nonNegativeIntegerOr(v: unknown): number | undefined { + const value = nonNegativeOr(v); + return value !== undefined && Number.isInteger(value) ? value : undefined; +} + +/** `null` is the one valid unknown-cost representation; undefined means invalid input. */ +function nullableCostOr(v: unknown): number | null | undefined { + if (v === null) return null; + return nonNegativeOr(v); +} + function stringOr(v: unknown): string | undefined { return typeof v === "string" && v !== "" ? v : undefined; } -/** Shapes a single run entry: score is the minimum requirement, other fields tolerate being absent; a bad entry returns null and is dropped. */ -function toRun(v: unknown): BenchmarkRunScore | null { - const r = asRecord(v); - const score = numberOr(r.score); - if (score === undefined) return null; - const cost = numberOr(r.cost); - const durationMs = numberOr(r.duration_ms); - const sessionId = stringOr(r.session_id); - return { - score, - ...(cost !== undefined ? { cost } : {}), - ...(durationMs !== undefined ? { durationMs } : {}), - ...(sessionId !== undefined ? { sessionId } : {}), - }; +function isWithin(parent: string, child: string): boolean { + const relative = path.relative(parent, child); + return relative !== "" && !relative.startsWith("..") && !path.isAbsolute(relative); } -/** Average of a metric across runs; undefined when there's no value at all (never forced to 0). */ -function averageOf(runs: BenchmarkRunScore[], pick: (r: BenchmarkRunScore) => number | undefined) { - const values = runs.map(pick).filter((v): v is number => v !== undefined); - if (values.length === 0) return undefined; - return values.reduce((a, b) => a + b, 0) / values.length; +function statementTitle(statement: string, fallback: string): string { + const heading = /^#\s+(.+)$/m.exec(statement)?.[1]?.trim(); + return heading?.replace(/^Case\s+\d+\s*:\s*/i, "") || fallback; +} + +async function readStatementTitle(readme: string, fallback: string): Promise { + const handle = await fs.open(readme, "r"); + try { + const buffer = Buffer.alloc(STATEMENT_TITLE_READ_BYTES); + const { bytesRead } = await handle.read(buffer, 0, buffer.length, 0); + return statementTitle(buffer.subarray(0, bytesRead).toString("utf8"), fallback); + } finally { + await handle.close(); + } +} + +/** Shapes one current-format Run; a malformed entry invalidates its containing Case. */ +function toRun(v: unknown): BenchmarkRunScore | null { + const r = asRecord(v); + const score = scoreOr(r.score); + const cost = nullableCostOr(r.cost); + const durationMs = nonNegativeIntegerOr(r.duration_ms); + const sessionId = stringOr(r.session_id); + if (score === undefined || cost === undefined || durationMs === undefined || !sessionId) + return null; + return { score, cost, durationMs, sessionId }; } /** - * Shapes a case-level entry (scoreboard v2): the three metrics trust the file's own - * values, falling back to an average over runs when missing; the old format (no - * runs, a single case-level session_id) is backfilled into a single run. case and a - * score (from the file or derivable from runs) are the minimum requirement, - * otherwise the entry is dropped. + * Shapes one current-format Case. Its stored aggregates are trusted as written: + * this parser intentionally performs no average or consistency calculation. */ function toCase(v: unknown): BenchmarkCaseScore | null { const cr = asRecord(v); const caseId = stringOr(cr.case); - if (caseId === undefined) return null; - const parsedRuns = Array.isArray(cr.runs) - ? cr.runs.map(toRun).filter((r): r is BenchmarkRunScore => r !== null) - : []; - const score = numberOr(cr.score) ?? averageOf(parsedRuns, (r) => r.score); - if (score === undefined) return null; - const cost = numberOr(cr.cost) ?? averageOf(parsedRuns, (r) => r.cost); - const durationMs = numberOr(cr.duration_ms) ?? averageOf(parsedRuns, (r) => r.durationMs); - const sessionId = stringOr(cr.session_id); - const runs: BenchmarkRunScore[] = - parsedRuns.length > 0 - ? parsedRuns - : [ - // The old format is parsed as a single run: the case-level values are that run's raw result. - { - score, - ...(cost !== undefined ? { cost } : {}), - ...(durationMs !== undefined ? { durationMs } : {}), - ...(sessionId !== undefined ? { sessionId } : {}), - }, - ]; + const score = scoreOr(cr.score); + const cost = nullableCostOr(cr.cost); + const durationMs = nonNegativeIntegerOr(cr.duration_ms); + if ( + !caseId || + score === undefined || + cost === undefined || + durationMs === undefined || + "max_score" in cr || + !Array.isArray(cr.runs) || + cr.runs.length === 0 + ) { + return null; + } + const parsedRuns = cr.runs.map(toRun); + if (parsedRuns.some((run) => run === null)) return null; + const runs = parsedRuns as BenchmarkRunScore[]; return { case: caseId, score, - ...(cost !== undefined ? { cost } : {}), - ...(durationMs !== undefined ? { durationMs } : {}), - ...(sessionId !== undefined ? { sessionId } : {}), + cost, + durationMs, runs, }; } -/** Shapes a single evaluation record: time and score are the minimum requirement, other fields (summary, etc.) tolerate being absent. */ +/** Shapes one current-format Evaluation and trusts its stored aggregate metrics. */ function toEvaluation(v: unknown): BenchmarkEvaluation | null { const r = asRecord(v); const time = r.time instanceof Date ? r.time.toISOString() : r.time; - const score = numberOr(r.score); - if (typeof time !== "string" || time === "" || score === undefined) return null; - const cases: BenchmarkCaseScore[] = Array.isArray(r.cases) - ? r.cases.map(toCase).filter((c): c is BenchmarkCaseScore => c !== null) - : []; + const score = scoreOr(r.score); + const cost = nullableCostOr(r.cost); + const durationMs = nonNegativeIntegerOr(r.duration_ms); const summary = stringOr(r.summary); // Title and body are separate: summary_title is a one-line // conclusion, summary is the body text. const summaryTitle = stringOr(r.summary_title); - // The Model actually used for this evaluation run (paired with provider): - // charted curves are split into series by model, each with a distinct color. const modelId = stringOr(r.model_id); const provider = stringOr(r.provider); - const version = numberOr(r.version); - const cost = numberOr(r.cost); - const durationMs = numberOr(r.duration_ms); + const thinkingLevel = stringOr(r.thinking_level); + const version = nonNegativeIntegerOr(r.version); + if ( + typeof time !== "string" || + time === "" || + score === undefined || + cost === undefined || + durationMs === undefined || + !modelId || + !provider || + !thinkingLevel || + version === undefined || + version < 1 || + !Array.isArray(r.cases) || + r.cases.length === 0 + ) { + return null; + } + const parsedCases = r.cases.map(toCase); + if (parsedCases.some((item) => item === null)) return null; + const cases = parsedCases as BenchmarkCaseScore[]; return { time, ...(summaryTitle !== undefined ? { summaryTitle } : {}), ...(summary !== undefined ? { summary } : {}), - ...(modelId !== undefined ? { modelId } : {}), - ...(provider !== undefined ? { provider } : {}), + modelId, + provider, + thinkingLevel, score, - ...(version !== undefined ? { version } : {}), - ...(cost !== undefined ? { cost } : {}), - ...(durationMs !== undefined ? { durationMs } : {}), + version, + cost, + durationMs, cases, }; } export class BenchmarkService { - constructor(private readonly root: string) {} + constructor( + private readonly root: string, + private readonly workspaceFiles: WorkspaceFilesService, + ) {} async list(projectId: string, agentId: string): Promise { const dir = benchmarksDir(this.root, projectId, agentId); @@ -157,6 +208,98 @@ export class BenchmarkService { return { benchmarks }; } + async listCases( + projectId: string, + agentId: string, + benchmarkId: string, + ): Promise { + const baseDir = benchmarksDir(this.root, projectId, agentId); + const benchDir = path.join(baseDir, benchmarkId); + let entries: Array<{ name: string; isDirectory(): boolean }>; + let realBaseDir: string; + let realBenchDir: string; + try { + [entries, realBaseDir, realBenchDir] = await Promise.all([ + fs.readdir(benchDir, { withFileTypes: true }), + fs.realpath(baseDir), + fs.realpath(benchDir), + ]); + } catch { + return { cases: [] }; + } + if (!isWithin(realBaseDir, realBenchDir)) return { cases: [] }; + + const cases: BenchmarkCaseSummary[] = []; + for (const entry of entries + .filter((item) => item.isDirectory() && item.name.startsWith("CASE-")) + .sort((a, b) => a.name.localeCompare(b.name))) { + const fallback: BenchmarkCaseSummary = { id: entry.name, title: entry.name }; + try { + const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, entry.name); + const realReadme = await fs.realpath(path.join(statementDir, "README.md")); + if (!isWithin(statementDir, realReadme)) throw new Error("README escapes Statement"); + cases.push({ + id: entry.name, + title: await readStatementTitle(realReadme, entry.name), + }); + } catch { + cases.push(fallback); + } + } + return { cases }; + } + + async listCaseFiles( + projectId: string, + agentId: string, + benchmarkId: string, + caseId: string, + rel: string, + ): Promise { + const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, caseId); + return this.workspaceFiles.list(statementDir, rel); + } + + async readCaseFile( + projectId: string, + agentId: string, + benchmarkId: string, + caseId: string, + rel: string, + options?: WorkspaceFileReadOptions, + ): Promise { + const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, caseId); + return this.workspaceFiles.read(statementDir, rel, options); + } + + private async statementRoot( + projectId: string, + agentId: string, + benchmarkId: string, + caseId: string, + ): Promise { + const benchDir = path.join(benchmarksDir(this.root, projectId, agentId), benchmarkId); + const caseDir = path.join(benchDir, caseId); + const statementDir = path.join(caseDir, "statement"); + try { + const [realBenchDir, realCaseDir, realStatementDir] = await Promise.all([ + fs.realpath(benchDir), + fs.realpath(caseDir), + fs.realpath(statementDir), + ]); + if ( + !isWithin(realBenchDir, realCaseDir) || + path.dirname(realStatementDir) !== realCaseDir || + path.basename(realStatementDir) !== "statement" + ) { + throw new Error("Statement path is not canonical"); + } + return realStatementDir; + } catch { + throw new HttpError(404, "not_found", "Public Case Statement does not exist."); + } + } + private async readBenchmark(benchDir: string, id: string): Promise { // benchmark_config.toml: title, description, and per-case run count (falls back // to defaults if corrupt). The model isn't part of the config — each evaluation @@ -195,11 +338,11 @@ export class BenchmarkService { // No scores yet. } - // Case count: number of case subfolders (the statement/rubric structure isn't validated here). + // Case count: number of semantic Case subfolders. let caseCount = 0; try { const entries = await fs.readdir(benchDir, { withFileTypes: true }); - caseCount = entries.filter((e) => e.isDirectory()).length; + caseCount = entries.filter((e) => e.isDirectory() && e.name.startsWith("CASE-")).length; } catch { // Stays at 0. } diff --git a/packages/server/src/services/workspace-files-service.ts b/packages/server/src/services/workspace-files-service.ts index e70d25d..5034c61 100644 --- a/packages/server/src/services/workspace-files-service.ts +++ b/packages/server/src/services/workspace-files-service.ts @@ -7,7 +7,7 @@ import fs from "node:fs/promises"; import { constants as fsc } from "node:fs"; import path from "node:path"; -import type { WorkspaceFilesResponse } from "../api/types.js"; +import type { WorkspaceFileEntry, WorkspaceFilesResponse } from "../api/types.js"; import { HttpError } from "../http/errors.js"; import { badRequest } from "../http/validate.js"; @@ -48,6 +48,13 @@ export interface WorkspaceFileContent { contentType: string; /** Types whose same-origin inline rendering would execute scripts (html/svg): inline preview must fall back to plain text. */ scriptable: boolean; + /** True when a bounded preview returned only the beginning of the file. */ + truncated?: boolean; +} + +export interface WorkspaceFileReadOptions { + /** Return at most this many bytes. Used for bounded text previews. */ + maxBytes?: number; } export class WorkspaceFilesService { @@ -181,9 +188,14 @@ export class WorkspaceFilesService { return unique.filter((_, i) => exists[i]); } - /** List a directory: dirs come first, each group sorted by name; kind follows the symlink target (consistent with read behavior). */ + /** + * List a directory: dirs come first, each group sorted by name. Entries whose + * canonical target leaves the Workspace are omitted, so listing cannot expose + * metadata for an out-of-bounds symlink. + */ async list(workspace: string, rel: string): Promise { const dir = await this.resolveRead(workspace, rel); + const realBase = await this.realBase(workspace); let dirents; try { dirents = await fs.readdir(dir, { withFileTypes: true }); @@ -196,28 +208,24 @@ export class WorkspaceFilesService { } throw err; } - const entries = await Promise.all( - dirents.map(async (d) => { - let sizeBytes = 0; - let mtime = ""; - // Dirent doesn't report the target type for a symlink, so stat (following the link) is used to determine dir/file. - let isDir = d.isDirectory(); + const listed = await Promise.all( + dirents.map(async (d): Promise => { try { - const stat = await fs.stat(path.join(dir, d.name)); - sizeBytes = stat.size; - mtime = stat.mtime.toISOString(); - isDir = stat.isDirectory(); + const canonical = await fs.realpath(path.join(dir, d.name)); + this.assertInside(canonical, realBase); + const stat = await fs.stat(canonical); + return { + name: d.name, + kind: stat.isDirectory() ? "dir" : "file", + sizeBytes: stat.size, + mtime: stat.mtime.toISOString(), + }; } catch { - // A dangling symlink or similar: keep the entry, with size/time left at defaults. + return null; } - return { - name: d.name, - kind: isDir ? ("dir" as const) : ("file" as const), - sizeBytes, - mtime, - }; }), ); + const entries = listed.filter((entry): entry is WorkspaceFileEntry => entry !== null); entries.sort((a, b) => a.kind === b.kind ? a.name.localeCompare(b.name) : a.kind === "dir" ? -1 : 1, ); @@ -225,7 +233,11 @@ export class WorkspaceFilesService { } /** Read a file (preview/download): IO on the canonical path (resolveRead has already eliminated symlink escapes). */ - async read(workspace: string, rel: string): Promise { + async read( + workspace: string, + rel: string, + options?: WorkspaceFileReadOptions, + ): Promise { const file = await this.resolveRead(workspace, rel); let stat; try { @@ -234,16 +246,38 @@ export class WorkspaceFilesService { throw new HttpError(404, "path_not_found", "File does not exist."); } if (stat.isDirectory()) throw badRequest("path is a directory."); - if (stat.size > MAX_READ_BYTES) { + const maxBytes = options?.maxBytes; + if ( + maxBytes !== undefined && + (!Number.isSafeInteger(maxBytes) || maxBytes < 1 || maxBytes > MAX_READ_BYTES) + ) { + throw badRequest("maxBytes must be a positive integer within the read limit."); + } + if (maxBytes === undefined && stat.size > MAX_READ_BYTES) { throw new HttpError(413, "file_too_large", "File exceeds the 50MB read limit."); } - const data = await fs.readFile(file); + let data: Buffer; + let truncated = false; + if (maxBytes !== undefined && stat.size > maxBytes) { + const handle = await fs.open(file, "r"); + try { + const buffer = Buffer.alloc(maxBytes); + const { bytesRead } = await handle.read(buffer, 0, maxBytes, 0); + data = buffer.subarray(0, bytesRead); + truncated = true; + } finally { + await handle.close(); + } + } else { + data = await fs.readFile(file); + } const ext = path.extname(file).toLowerCase(); return { data, fileName: path.basename(file), contentType: CONTENT_TYPES[ext] ?? "application/octet-stream", scriptable: ext === ".html" || ext === ".htm" || ext === ".svg", + ...(truncated ? { truncated: true } : {}), }; } diff --git a/packages/server/test/benchmarks.test.ts b/packages/server/test/benchmarks.test.ts index fc45865..e0dc56a 100644 --- a/packages/server/test/benchmarks.test.ts +++ b/packages/server/test/benchmarks.test.ts @@ -1,10 +1,9 @@ /** * Benchmark scoreboard read integration tests (read-only display): benchmark_config.toml title/description and runs * pass-through (falls back to directory name if missing), scoreboard.yaml v2's - * evaluations[] (summary pass-through, per-case runs array; per-case metrics trust the - * file values, falling back to an average over runs when missing), the legacy format - * (per-case single session_id) parsed as a single run and backfilled, bad entries - * discarded, case count, empty when unconfigured, permissions (members can read, + * evaluations[] (summary pass-through, model-written Case/Evaluation averages and per-case + * runs arrays), rejection of legacy Scoreboard entries, case count, empty when + * unconfigured, permissions (members can read, * outsiders get 404). * * Tested with a plain Agent (no sample Benchmark pre-installed); default_agent's sample @@ -14,7 +13,12 @@ import fs from "node:fs/promises"; import path from "node:path"; import { afterEach, beforeEach, describe, expect, it } from "vitest"; import { benchmarksDir } from "@prismshadow/penguin-core"; -import type { BenchmarksResponse, ProjectCreateResponse } from "../src/api/types.js"; +import type { + BenchmarkCasesResponse, + BenchmarksResponse, + ProjectCreateResponse, + WorkspaceFilesResponse, +} from "../src/api/types.js"; import { apiClient, createTestApp, provisionUser } from "./helpers.js"; import type { TestApp } from "./helpers.js"; @@ -59,10 +63,48 @@ describe("benchmarks api", () => { }); }); - it("scoreboard v2: summary/runs pass through; metrics trust file or average runs", async () => { + it("current scoreboard: model-written averages, runtime, and runs pass through", async () => { const dir = path.join(benchmarksDir(t.root, projectId, AGENT), "swe-bench-v2"); await fs.mkdir(path.join(dir, "CASE-001-excel-task", "statement"), { recursive: true }); + await fs.mkdir(path.join(dir, "CASE-001-excel-task", "statement", "assets"), { + recursive: true, + }); + await fs.mkdir(path.join(dir, "CASE-002-web-task", "statement"), { recursive: true }); await fs.mkdir(path.join(dir, "CASE-002-web-task", "rubric"), { recursive: true }); + await fs.writeFile( + path.join(dir, "CASE-001-excel-task", "statement", "README.md"), + "# Case 001: Excel cleanup\n\nClean the workbook.", + "utf8", + ); + await fs.writeFile( + path.join(dir, "CASE-001-excel-task", "statement", "data.csv"), + "id,value\n1,alpha\n", + "utf8", + ); + await fs.writeFile( + path.join(dir, "CASE-001-excel-task", "statement", "assets", "notes.txt"), + "public notes", + "utf8", + ); + await fs.writeFile( + path.join(dir, "CASE-001-excel-task", "statement", "large.txt"), + "x".repeat(300 * 1024), + "utf8", + ); + await fs.writeFile( + path.join(dir, "CASE-002-web-task", "statement", "README.md"), + "# Web task\n\nBuild the page.", + "utf8", + ); + await fs.writeFile( + path.join(dir, "CASE-002-web-task", "rubric", "README.md"), + "PRIVATE GOLD: never return this text", + "utf8", + ); + await fs.symlink( + path.join(dir, "CASE-002-web-task", "rubric", "README.md"), + path.join(dir, "CASE-001-excel-task", "statement", "private-link.md"), + ); await fs.writeFile( path.join(dir, "benchmark_config.toml"), `title = "SWE Bench v2"\ndescription = "Example"\nruns = 2\n`, @@ -76,35 +118,39 @@ describe("benchmarks api", () => { " version: 3", ' provider: "deepseek"', ' model_id: "deepseek-v4-pro"', + ' thinking_level: "medium"', ' summary_title: "Added planning steps to the system Prompt"', ' summary: "Each case run twice and averaged; added planning steps."', - " score: 8.0", - " cost: 0.05", - " duration_ms: 120000", + " score: 72.35", + " cost: 0.04", + " duration_ms: 42500", " cases:", - // Per-case metrics are all present: trust the file values (no recomputation even if inconsistent with the runs average). + // Stored averages are authoritative even when inconsistent with the raw Runs. ' - case: "CASE-001-excel-task"', - " score: 4.2", - " cost: 0.02", + " score: 80.2", + " cost: 0.04", " duration_ms: 50000", " runs:", - " - score: 4.0", - " cost: 0.018", + " - score: 80", + " cost: null", " duration_ms: 48000", ' session_id: "session-run-1"', - " - score: 4.5", - " cost: 0.022", + " - score: 82", + " cost: 0.04", " duration_ms: 52000", ' session_id: "session-run-2"', - // Per-case metrics are missing: computed as the average over runs. + // All unknown Run costs produce a model-written null Case cost; the Evaluation ignores it. ' - case: "CASE-002-web-task"', + " score: 64.5", + " cost: null", + " duration_ms: 35000", " runs:", - " - score: 3.0", - " cost: 0.01", + " - score: 60", + " cost: null", " duration_ms: 30000", ' session_id: "session-run-3"', - " - score: 4.0", - " cost: 0.03", + " - score: 70", + " cost: null", " duration_ms: 40000", ' session_id: "session-run-4"', ].join("\n"), @@ -127,28 +173,100 @@ describe("benchmarks api", () => { // The evaluation entry carries this run's model (as a pair) and a summary title (curve series / title-body are displayed separately). expect(evaluation.provider).toBe("deepseek"); expect(evaluation.modelId).toBe("deepseek-v4-pro"); + expect(evaluation.thinkingLevel).toBe("medium"); expect(evaluation.summaryTitle).toBe("Added planning steps to the system Prompt"); expect(evaluation.summary).toBe("Each case run twice and averaged; added planning steps."); - expect(evaluation.score).toBe(8.0); - // Per-case metrics are all present: trust the file (4.2, not the runs average of 4.25). + expect(evaluation.score).toBe(72.35); + expect(evaluation.cost).toBe(0.04); + expect(evaluation.durationMs).toBe(42500); + expect("maxScore" in evaluation).toBe(false); + // Per-case metrics trust the file (80.2, not the Runs' arithmetic mean of 81). const full = evaluation.cases.find((c) => c.case === "CASE-001-excel-task")!; - expect(full.score).toBe(4.2); - expect(full.cost).toBe(0.02); + expect(full.score).toBe(80.2); + expect(full.cost).toBe(0.04); expect(full.durationMs).toBe(50000); expect(full.runs).toEqual([ - { score: 4.0, cost: 0.018, durationMs: 48000, sessionId: "session-run-1" }, - { score: 4.5, cost: 0.022, durationMs: 52000, sessionId: "session-run-2" }, + { score: 80, cost: null, durationMs: 48000, sessionId: "session-run-1" }, + { score: 82, cost: 0.04, durationMs: 52000, sessionId: "session-run-2" }, ]); - // Per-case metrics are missing: computed as the average over runs. - const derived = evaluation.cases.find((c) => c.case === "CASE-002-web-task")!; - expect(derived.score).toBe(3.5); - expect(derived.cost).toBeCloseTo(0.02, 10); - expect(derived.durationMs).toBe(35000); - expect(derived.runs).toHaveLength(2); - expect(derived.sessionId).toBeUndefined(); + const partialCost = evaluation.cases.find((c) => c.case === "CASE-002-web-task")!; + expect(partialCost.score).toBe(64.5); + expect(partialCost.cost).toBeNull(); + expect(partialCost.durationMs).toBe(35000); + expect(partialCost.runs).toEqual([ + { score: 60, cost: null, durationMs: 30000, sessionId: "session-run-3" }, + { score: 70, cost: null, durationMs: 40000, sessionId: "session-run-4" }, + ]); + + const caseResponse = (await ( + await member.get(`${base}/swe-bench-v2/cases`) + ).json()) as BenchmarkCasesResponse; + expect(caseResponse).toEqual({ + cases: [ + { + id: "CASE-001-excel-task", + title: "Excel cleanup", + }, + { + id: "CASE-002-web-task", + title: "Web task", + }, + ], + }); + expect(JSON.stringify(caseResponse)).not.toContain("PRIVATE GOLD"); + + const filesBase = `${base}/swe-bench-v2/cases/CASE-001-excel-task/files`; + const files = (await (await member.get(filesBase)).json()) as WorkspaceFilesResponse; + expect(files.path).toBe(""); + expect(files.entries.map((entry) => `${entry.kind}:${entry.name}`)).toEqual([ + "dir:assets", + "file:data.csv", + "file:large.txt", + "file:README.md", + ]); + expect(JSON.stringify(files)).not.toContain("private-link.md"); + expect( + ( + await member.get( + `${filesBase}/content?path=${encodeURIComponent("private-link.md")}&preview=1`, + ) + ).status, + ).toBe(400); + + const nested = (await ( + await member.get(`${filesBase}?path=${encodeURIComponent("assets")}`) + ).json()) as WorkspaceFilesResponse; + expect(nested.entries.map((entry) => entry.name)).toEqual(["notes.txt"]); + + const statement = await member.get( + `${filesBase}/content?path=${encodeURIComponent("README.md")}&preview=1`, + ); + expect(statement.status).toBe(200); + expect(statement.headers.get("content-type")).toContain("markdown"); + expect(await statement.text()).toBe("# Case 001: Excel cleanup\n\nClean the workbook."); + + const large = await member.get( + `${filesBase}/content?path=${encodeURIComponent("large.txt")}&preview=1`, + ); + expect(large.status).toBe(200); + expect(large.headers.get("x-content-truncated")).toBe("1"); + expect((await large.text()).length).toBe(256 * 1024); + + const download = await member.get( + `${filesBase}/content?path=${encodeURIComponent("data.csv")}&download=1`, + ); + expect(download.headers.get("content-disposition")).toContain("attachment"); + expect(await download.text()).toBe("id,value\n1,alpha\n"); + + expect( + (await member.get(`${filesBase}/content?path=${encodeURIComponent("../rubric/README.md")}`)) + .status, + ).toBe(400); + expect((await outsider.get(`${base}/swe-bench-v2/cases`)).status).toBe(404); + expect((await outsider.get(filesBase)).status).toBe(404); }); - it("legacy per-case session_id parsed as one backfilled run; bad entries dropped", async () => { + it("does not migrate or backfill legacy Scoreboard entries", async () => { const dir = path.join(benchmarksDir(t.root, projectId, AGENT), "swe-bench-v1"); await fs.mkdir(path.join(dir, "CASE-001-excel-task", "statement"), { recursive: true }); await fs.writeFile(path.join(dir, "benchmark_config.toml"), `title = "SWE Bench v1"\n`, "utf8"); @@ -184,26 +302,7 @@ describe("benchmarks api", () => { const bench = res.benchmarks[1]!; expect(bench).toMatchObject({ title: "SWE Bench v1", caseCount: 1 }); expect("runs" in bench).toBe(false); - expect(bench.evaluations).toHaveLength(1); - expect(bench.evaluations[0]).toMatchObject({ - time: "2026-07-16T10:00:00Z", - version: 1, - score: 62.5, - cost: 1.25, - durationMs: 60000, - }); - expect("summary" in bench.evaluations[0]!).toBe(false); - // Legacy per-case format: fields unchanged, plus one backfilled run matching the case-level values (the frontend uniformly expands via runs). - expect(bench.evaluations[0]?.cases).toEqual([ - { - case: "CASE-001-excel-task", - score: 30, - cost: 0.5, - durationMs: 20000, - sessionId: "session-abc", - runs: [{ score: 30, cost: 0.5, durationMs: 20000, sessionId: "session-abc" }], - }, - ]); + expect(bench.evaluations).toEqual([]); expect(res.benchmarks[0]).toMatchObject({ title: "empty-bench", caseCount: 0, diff --git a/packages/server/test/builtin-agents.test.ts b/packages/server/test/builtin-agents.test.ts index 7a9c3fc..73af9df 100644 --- a/packages/server/test/builtin-agents.test.ts +++ b/packages/server/test/builtin-agents.test.ts @@ -88,6 +88,7 @@ describe("built-in Agent provisioning", () => { expect(bench.caseCount).toBe(2); expect(bench.evaluations).toHaveLength(3); for (const evaluation of bench.evaluations) { + expect(evaluation.thinkingLevel).toBe("medium"); expect(evaluation.summary).toBeTruthy(); expect(evaluation.cases).toHaveLength(2); for (const c of evaluation.cases) { diff --git a/packages/server/test/workspace-files.test.ts b/packages/server/test/workspace-files.test.ts index d99e995..305c8bd 100644 --- a/packages/server/test/workspace-files.test.ts +++ b/packages/server/test/workspace-files.test.ts @@ -45,6 +45,9 @@ describe("workspace-files-service", () => { const file = await svc.read(ws, "sub/c.md"); expect(file.data.toString()).toBe("# md"); expect(file.contentType).toContain("markdown"); + const preview = await svc.read(ws, "sub/c.md", { maxBytes: 2 }); + expect(preview.data.toString()).toBe("# "); + expect(preview.truncated).toBe(true); await expect(svc.read(ws, "sub")).rejects.toMatchObject({ status: 400 }); await expect(svc.read(ws, "nope.txt")).rejects.toMatchObject({ status: 404 }); }); @@ -86,6 +89,8 @@ describe("workspace-files-service", () => { it("symlink escape: reads and writes are both rejected when the link points outside the Workspace", async () => { await fs.symlink(outside, path.join(ws, "link-out")); + const root = await svc.list(ws, ""); + expect(root.entries.map((entry) => entry.name)).not.toContain("link-out"); await expect(svc.list(ws, "link-out")).rejects.toMatchObject({ status: 400 }); await expect(svc.read(ws, "link-out/secret.txt")).rejects.toMatchObject({ status: 400 }); // Writing outside via a directory symlink: caught by the parent-directory realpath check. diff --git a/packages/skills/skills/agent-creation/SKILL.md b/packages/skills/skills/agent-creation/SKILL.md index 97a91a8..edcb1e7 100644 --- a/packages/skills/skills/agent-creation/SKILL.md +++ b/packages/skills/skills/agent-creation/SKILL.md @@ -1,10 +1,10 @@ --- name: agent-creation -description: Turn a user requirement into a concrete agent — write the target agent's AGENTS.md and install the skills it needs. +description: Create or configure an Agent State from a user requirement by writing AGENTS.md, setting identity metadata, and installing only needed Skills. short_description: Turn a requirement into a working agent. short_description_zh: 把需求变成可用的 Agent。 -version: 5 -updated: 2026-07-25T00:00:00Z +version: 6 +updated: 2026-07-29T17:20:58Z --- # Agent Creation @@ -15,6 +15,15 @@ This skill turns a user requirement into a working agent configuration — plain If the user's message only invokes this skill (e.g. "use agent-creation skill") without a concrete requirement, ask the user what agent they want and what it should do. But when the requirement is already concrete — even a single sentence like "an expert that answers questions about X" — do **not** ask follow-up questions: derive the role and rules from that sentence, apply the defaults below, and list your assumptions in the final reply. +## Resolve the inherited runtime + +Treat the current Agent as the **Builder**. Resolve the runtime before creating a new Agent: + +- `provider` and `model_id` are one complete pair. If the user explicitly supplies both, use that pair. If the user supplies neither, inherit the current Builder Session's `Provider` and `Model ID` from the Environment. Reject a half pair. +- `thinking_level` is independent. If the user explicitly supplies it, use that value. Otherwise read `model.thinking_level` from the Builder's own `agent_state/system_config.yaml`; when the field is absent, use the normal Agent-config default `medium`. + +Write the resolved `thinking_level` into a brand-new target Agent's `model.thinking_level`, preserving all other copied `model` fields. Penguin does not persist `provider` or `model_id` in Agent State, so never add either field to `system_config.yaml`. When the same request continues into Benchmark design, carry the resolved model pair forward explicitly so evaluation uses the Builder runtime instead of a Project default. When configuring an existing Agent, change `model.thinking_level` only when the user explicitly requests that runtime change. + ## Locate the target agent All agents of this project live side by side under `agents/` in the App Data Dir: @@ -63,20 +72,35 @@ Library skills can be copied from any agent that already has them (e.g. `default - **Knowledge expert** (answers questions over a document set): usually **no** harness agent is needed — build a RAG app with the penguin-sdk skill instead, and configure the app's embedded agent (below). - **Evaluation loop**: `benchmark-design`, `agent-evaluation`, `agent-optimization`. +When creating a Test Agent, install only the capabilities it needs to solve ordinary tasks. + ## Set name and description -In the target's `agent_state/system_config.yaml`, set the top-level `name:` and `description:` fields so the agent is recognizable in lists. Edit only these two fields. +In the target's `agent_state/system_config.yaml`, set the top-level `name:` and `description:` fields so the agent is recognizable in lists. For an existing Agent, edit only these two fields unless the user explicitly requested a `thinking_level` change. ## Creating a brand-new agent -Prefer configuring an agent the user already created. If you must create one from scratch: pick a short id (letters, digits, `_`, `-`), copy the default agent's `system_config.yaml` as the base, and create the layout described above: +Prefer configuring an agent the user already created. If the user requires a new Agent, confirm that `TARGET` does not exist. If it already exists, stop and tell the user; never silently overwrite, reinitialize, or reuse an existing Agent under the same id. + +After confirming that the target is absent, pick a short id using letters, digits, `_`, or `-`, copy the default Agent's `system_config.yaml` as the base, and create the layout described above: ```bash mkdir -p "$TARGET/agent_state/skills" "$TARGET/agent_state/memory" "$TARGET/agent_state/tools" "$TARGET/scratchpad" cp "$APP_DATA_DIR/agents/default_agent/agent_state/system_config.yaml" "$TARGET/agent_state/" ``` -A new agent starts with no skills — install only what it needs. Then write its AGENTS.md, name and description as above. +Then set the top-level `name`, `description`, and `version: 1`, set `model.thinking_level` to the resolved value, write `AGENTS.md`, and install only the Skills required by the user's requirement. Do not persist the resolved provider/model pair in the Agent State. + +## Validate and report + +Before finishing: + +- parse `agent_state/system_config.yaml` and confirm `name`, `description`, a positive integer `version`, and the expected `model.thinking_level`; +- confirm `agent_state/AGENTS.md` exists and is non-empty; +- confirm every installed Skill has a parseable `SKILL.md`, and its `name` matches its directory; +- confirm no Agent outside `TARGET` was changed. + +Report the target path, whether an existing Agent was configured or a new Agent was created, assumptions, installed Skills, the resolved runtime and whether each value was user-specified or inherited, and validation results. ## The embedded agent of an SDK app diff --git a/packages/skills/skills/agent-evaluation/SKILL.md b/packages/skills/skills/agent-evaluation/SKILL.md index a826788..74e6b5c 100644 --- a/packages/skills/skills/agent-evaluation/SKILL.md +++ b/packages/skills/skills/agent-evaluation/SKILL.md @@ -1,25 +1,27 @@ --- name: agent-evaluation -description: Run and score exactly one Benchmark Case run with CLI execution, Trace provenance checks, and private Rubric isolation. +description: Run one specified Test Agent on one specified Benchmark Case exactly once, privately score that execution, and return one protocol result. short_description: Run and score one isolated Benchmark Case. short_description_zh: 隔离执行并评分一个 Benchmark Case。 -version: 4 -updated: 2026-07-25T00:00:00Z +version: 5 +updated: 2026-07-29T17:20:58Z --- # Agent Evaluation -Act as an internal leaf worker. For one valid request, run and score exactly one Benchmark Case once, then return minimal protocol metadata. Do not design or refine the Benchmark, modify the Test Agent State, or write `scoreboard.yaml`. Do not use `run_subagent` or `input_subagent`. +Handle one evaluation request from a `run_subagent` caller: run the specified Test Agent on one Benchmark Case once, score that execution privately, and return one protocol result. + +The top-level Benchmark Designer or Optimizer owns all Case and Run loops, concurrency, and follow-up handling. This worker handles no other Case or Run, launches no evaluator or subagent, modifies no Agent or Benchmark, and never writes `scoreboard.yaml`. Use the Penguin CLI only to launch the specified Test Agent; do not use it to create another phase, designer, optimizer, or evaluator. + +Operate silently. Call tools without progress messages. Across all streamed and final responses, the only worker-authored text must be the final plain protocol YAML. Emit no narration, headings, Markdown fences, summaries, private scoring details, or other text. ## Before you start -This Skill is invoked by `benchmark-design` or Benchmark mode in `agent-optimization`. Require one unambiguous protocol request containing every identity field below. If the request is missing, duplicated, or conflicting, return `invalid_request` without creating a Workspace or launching the Test Agent. Do not ask an interactive clarification from this leaf worker. +Use this Skill only for a complete request from a `run_subagent` caller. If the request is incomplete or inconsistent, return `invalid_request` through the protocol instead of asking the user a question. -## Privacy boundary +## Contract -Before the final protocol YAML, emit no assistant text; use private reasoning and tool calls only. Never serialize Statement or artifact contents, Rubric items, expected values, correct outcomes, per-item scoring, diagnostics, secret configuration, Workspace paths, or Trace paths into an assistant message. The final assistant message is the protocol YAML only. It may echo the public identity fields supplied by the caller. - -A valid request contains exactly one value for each field: +Require exactly one value for every field below: ```text protocol_version: 1 @@ -27,71 +29,74 @@ case_id: run: <1_based_run_index> expected_version: test_agent_id: -benchmark_dir: +benchmark_id: provider: -model_id: +model_id: ``` -## Validate and prepare +One request represents one Test Agent execution. The `run` value identifies that execution; it is not a repeat count. `provider` and `model_id` must both be non-empty and select that exact configured model. If a required field is missing, duplicated, or conflicting, return `invalid_request` without creating a Workspace or launching the Test Agent. + +Return a **scored result** when the Test Agent ran and the Rubric could be applied. Wrong, malformed, or missing Test Agent output is still a scored result. Return an **evaluation failure** when the request, Benchmark, launch, version check, Trace binding, or scoring process prevents a valid score. Resolve the Project, Test Agent, Benchmark, and Case only from the explicit request and Environment App Data Dir. Reject traversal, symlink escape, or any path outside the requested Test Agent. Never read a Project configuration file, credential, or vault. -Require: +## Prepare + +Use the `App Data Dir` from the Environment: ```text -/agent_state/system_config.yaml -/benchmark_config.toml -//statement/README.md -//rubric/README.md +TEST_AGENT_DIR = /agents/ +BENCHMARK_DIR = /benchmarks/ ``` -Require `benchmark_config.toml` to contain a positive integer `runs`; the requested `run` must be within `1..runs`. The canonical State version is the top-level `version` in `system_config.yaml`, defaulting to 1, and must equal `expected_version`. +Reject path traversal, symlink escape, or any resolved path outside the requested Test Agent. Inspect only the requested Agent State, Benchmark config and Case, isolated Test Workspace, and Traces needed to verify this execution. Do not inspect another Agent, Project secrets, hidden configuration, or unrelated Workspaces or Traces. -Read and retain the exact Statement and Rubric bytes before launch. Reject an unusable, contradictory, non-atomic, or unbounded Rubric. The Rubric must declare a finite Case maximum; the returned score must fall within `0..case_max`. +Require `agent_state/system_config.yaml`, `benchmark_config.toml`, `/statement/README.md`, and `/rubric/README.md`. Check that `runs` is positive and `run` is within `1..runs`. The top-level Agent State `version`, defaulting to 1, must equal `expected_version`; otherwise return `version_changed`. Read and snapshot `model.thinking_level` from this Target Agent config, using the normal Agent-config default `medium` only when the field is absent. This configured value is the evaluation `thinking_level`; do not require or read thinking metadata from a Trace. -Create a collision-checked Workspace at `/workspaces/tmp-<8hex>`. Copy only the contents of `statement/` into it. Never copy, link, or disclose `rubric/`, and never reuse another Case or run's Workspace. +Before launch, snapshot every file under the Case's `statement/` and `rubric/` directories. Require a usable Rubric whose scoring items total exactly 100 points. Create a unique Workspace under `/workspaces/`, resolve it to an absolute canonical path, and verify that the resolved path remains under that directory. Copy only `statement/` into it. The Test Agent may see the Statement and its own State, but never the Rubric, Gold answers, scoring rules, or Evaluator reasoning. -## Launch and bind the Test Session +## Run and verify -Use an existing verified Penguin CLI or repository-local launcher already available in the runtime. Do not install a CLI and do not use `penguin run` as a probe. If no launcher is available, return `cli_failed`. +Use an existing verified Penguin CLI or repository-local launcher. Do not install or probe a launcher. Snapshot the isolated Workspace and record the existing Trace files. -Run the Test Agent exactly once in the foreground with a fresh top-level Session: +Resolve `PROJECT_DIR`, then derive and verify `PROJECT_ID`, then derive and verify `PENGUIN_HOME`. Perform these as separate shell statements in this order. Never compress the assignments onto one command line, derive a value before its input exists, or substitute another Penguin home. Before launch, confirm that `PROJECT_ID` equals the basename of `PROJECT_DIR` and `PENGUIN_HOME` equals its dirname. + +Start one foreground execution with a fresh top-level Session. With an explicit pair, use: ```bash PROJECT_DIR="" # the App Data Dir value from your Environment section is the project root PROJECT_ID="$(basename "$PROJECT_DIR")" PENGUIN_HOME="$(dirname "$PROJECT_DIR")" -WORKSPACE="$PROJECT_DIR/agents//workspaces/" export PENGUIN_HOME -penguin run --message "Read README.md in the current Workspace and complete the task exactly as specified there." \ +penguin run \ + --message "Read README.md in the current Workspace and complete the task exactly as specified there." \ --provider "" --model-id "" --project-id "$PROJECT_ID" \ - --agent-id "" --workspace "$WORKSPACE" --approve allow-all + --agent-id "" --workspace "" --approve allow-all ``` -Use the exact Project, Test Agent, Model pair, and Workspace. Do not fall back to another value. Poll the same process until it exits. A nonzero, interrupted, or misrouted launch is `cli_failed`, not score zero. Do not relaunch within the same run or target processes by a global name or pattern. +Use the exact requested Agent, Project, absolute Workspace path, and model pair. Never omit either model flag and never fall back to a Project default. If a launch fails, retry only when unchanged Workspace and Trace evidence proves that the Test Agent did not start. Every retry must follow a new diagnosis and apply a specific correction; never repeat an unchanged launch. Do not impose a numeric retry limit while distinct safe repairs remain. Return `evaluation_failed` when no new repair remains, external configuration is required, or it is unclear whether the Test Agent started. -Read the canonical State version before and after the Test run; any change is `version_changed`. Confirm the Statement and Rubric bytes are unchanged before scoring. +Verify after the run that the State version, configured `model.thinking_level`, and both directory snapshots are unchanged. Return `version_changed` when the State version or configured thinking level differs and `benchmark_invalid` when the Statement or Rubric differs. -Search only the explicitly requested Test Agent's `traces/` tree and never inspect another Agent's traces. Group rotated shards by Session. Evaluate every Session group that could contain the exact Workspace match; never infer ownership from recency or a fixed-size latest subset. Bind the Test Trace mechanically from `session_meta`: +Inspect only new or changed Traces. Bind exactly one root Test Trace whose Workspace, Agent State path, provider, and model match this request. Ignore unrelated parallel Traces and exclude the root Trace's directly referenced child Sessions. Return `evaluation_failed` if there is no unique match. Read the actual non-empty `provider` and `model_id` from the bound root Trace's `session_meta`; return `evaluation_failed` if either is unavailable. Use the unchanged Target Agent configuration snapshot—not Trace metadata—for `thinking_level`. -- `payload.workspace` equals the unique Workspace; -- `payload.agent_state` equals the exact Test Agent State path; -- `payload.provider` equals the requested provider; -- `payload.model_id` equals the requested model id. +## Score -When matching Test subagents exist, exclude child ids referenced by subagent events and require one unique matching root Test Session. Unrelated concurrent traces are not conflicts. Missing, multiple, malformed, or identity-mismatched roots are `provenance_mismatch`. +Inspect only the isolated Workspace, the bound root Trace, its directly referenced child Traces, and the private Rubric. Apply every scoring item and allowed equivalent. Keep Rubric contents, Gold answers, per-item scoring, and scoring rationale private. -## Score and account +A wrong answer, missing artifact, malformed output, or task failure attributable to the Test Agent is scored behavior and returns `status: ok`. A launcher, Trace-binding, or Evaluator failure is not scored. Return `benchmark_invalid` when the Rubric cannot be applied and `evaluation_failed` when the score is non-finite or outside `0..100`. -Inspect only the unique Test Workspace, its bound Test Trace, and the retained private Rubric. Apply every atomic item exactly and normalize only allowed equivalents. A missing, malformed, wrong-type, or incorrect Test artifact is ordinary scored Test Agent behavior: apply the Rubric's zero or partial credit and return `status: ok`. Only a changed or unusable Rubric, or a non-finite/out-of-range result, is `invalid_score`. Detailed reasoning remains in Evaluator Trace. +Set `duration_ms` from the root Test Session. Compute cost only from reliable final cumulative usage or cost already recorded in that Session and directly referenced child Traces found in the same bounded pass. Never browse, query a pricing service, or infer cost from external model prices. If the required data is unavailable, return `cost: null`. Missing cost data must not invalidate a score. -Compute `duration_ms` from the bound root Test Session, not from the Evaluator. Compute cost only from that root and child Sessions mechanically referenced by subagent events whose traces are available within the explicitly requested Test Agent's `traces/` tree. For each included Session, use final cumulative token usage rather than summing intermediate cumulative events, and apply the matching public `(provider, model_id)` pricing. If any referenced child trace or usage is unavailable there, including because the Test Session delegated to another Agent, or if any included usage is unpriced, return `cost: null` rather than inspecting another Agent or reporting a known partial sum as complete cost. +Round `score` to two decimal places. Preserve a non-null `cost` at the precision recorded in the Trace; do not round it. Write `duration_ms` as a non-negative integer rounded to the nearest millisecond. -## Return protocol +## Return -Emit exactly one plain YAML document beginning with `protocol_version:` and stop. Do not use a code fence or add explanations. +Return the required YAML as the only worker-authored text. Do not wrap it in backticks or a Markdown fence. -On success: +If the caller reports that your response formatting was invalid, use the scored or failed result already present in this Session and resend only the clean protocol YAML. Do not call tools, relaunch the Test Agent, rescore, or add an explanation. + +For a scored result: ```text protocol_version: 1 @@ -99,25 +104,34 @@ status: ok case_id: run: expected_version: -provider: -model_id: -score: <0_to_case_max> +provider: +model_id: +thinking_level: +score: <0_to_100> cost: duration_ms: session_id: ``` -On failure: +For an evaluation failure, use `null` for an identity field that was missing or conflicting: ```text protocol_version: 1 -status: infrastructure_failure -case_id: -run: -expected_version: -provider: -model_id: +status: failed +case_id: +run: +expected_version: +provider: +model_id: +thinking_level: failure_code: ``` -Stable codes are `invalid_request`, `invalid_statement`, `invalid_rubric`, `cli_failed`, `provenance_mismatch`, `version_changed`, and `invalid_score`. Do not include score, cost, duration, Session id, private data, or optimization advice on failure. +Use four failure codes: + +- `invalid_request`: the request is incomplete or inconsistent. +- `benchmark_invalid`: the Statement, Rubric, or scoring contract is invalid. +- `version_changed`: the Test Agent version does not match the request or changed during evaluation. +- `evaluation_failed`: launch could not be safely repaired, or Trace binding or scoring failed. + +Never include score, cost, duration, Session id, private data, or optimization advice on failure. diff --git a/packages/skills/skills/agent-optimization/SKILL.md b/packages/skills/skills/agent-optimization/SKILL.md index 47eb198..284fbc4 100644 --- a/packages/skills/skills/agent-optimization/SKILL.md +++ b/packages/skills/skills/agent-optimization/SKILL.md @@ -1,25 +1,29 @@ --- name: agent-optimization -description: Improve an Agent State from direct feedback or versioned multi-Case Benchmark scores and score-linked Traces. -short_description: Improve an Agent from feedback or measured Benchmark results. -short_description_zh: 根据反馈或 Benchmark 结果改进 Agent。 -version: 6 -updated: 2026-07-29T00:00:00Z +description: Improve an Agent State through versioned scores and score-linked Traces from a frozen Benchmark. +short_description: Improve an Agent from measured Benchmark results. +short_description_zh: 根据 Benchmark 结果改进 Agent。 +version: 7 +updated: 2026-07-30T02:51:10Z --- # Agent Optimization -Improve an existing Agent State. Use one-shot feedback mode for a direct correction, or Benchmark optimization mode for a measured loop. Do not mix their execution paths. One-shot mode does not require Agent Evaluator. Benchmark mode evaluates every configured Case run through Agent Evaluator and never launches or scores the Test Agent directly. +Improve one Test Agent through an evidence → hypothesis → Candidate → evaluation → accept or rollback loop. Use public Statements, scores, and Test Traces as black-box feedback. Delegate every evaluation to an `agent-evaluation` subagent; never run or score the Test Agent directly. ## Before you start -If the request supplies neither concrete feedback nor an explicit Test Agent and Benchmark, ask what to improve. Determine the mode before editing anything. +If the request does not identify the Test Agent, frozen Benchmark, desired target score, and round limit, ask for the missing inputs. When they are already supplied, proceed without asking the user to restate them. -Benchmark mode requires a top-level Session with `run_subagent`, a complete baseline series in `scoreboard.yaml`, and `agent-evaluation` installed on the current Agent. If any requirement is missing, stop and explain what the user must provide or install. Do not begin a partial optimization round. +## Goal and contract -## Pick the target Agent +Require an explicit Test Agent, a frozen Benchmark with a complete valid Formal Baseline, a desired target score, and a positive round limit. Read the evaluation `(provider, model_id, thinking_level)` from the complete Evaluation that matches the current Agent State; do not require the user to repeat it. An Evaluation without any part of this runtime is incomplete and cannot be used as a Reference. The top-level Session must provide `run_subagent`, and the current Agent must have the `agent-evaluation` Skill. If a prerequisite is missing, stop and explain what is needed. Do not create the missing Agent, Benchmark, or Baseline, and do not evaluate the Test Agent directly. -A one-shot request normally names the target. A delegated request begins with `Caller agent: `, and an `/agent` handoff contains `[handoff_from]`. When one-shot mode has no explicit target, use that caller or origin; if neither exists, ask. Benchmark mode always requires an explicit Test Agent and Benchmark. +A **Reference** is the Agent State currently kept as best, together with its complete Evaluation on the frozen Benchmark. + +Each round starts from the Reference and tests a bounded, general **Candidate**. Evaluate every Candidate on the same frozen Case × Run matrix and evaluation runtime. Accept it only when the change is admissible, the matrix is complete and valid, and its Evaluation's top-level `score` is strictly higher than the Reference Evaluation's `score`. An accepted Candidate and its Evaluation become the next Reference; otherwise restore the previous Reference. Stop early when the Reference reaches the desired target; otherwise run no more than the requested number of complete valid Candidate rounds. + +## Access and changes Resolve paths from the Environment's App Data Dir without recursively discovering the Project: @@ -32,115 +36,93 @@ STATE = /agent_state TRACES = /traces BENCHMARK = /benchmarks/ SCOREBOARD = /scoreboard.yaml +SNAPSHOTS = /snapshots ``` -Never read a Project configuration file, credential, vault, private Rubric, Agent Evaluator State, Evaluator Workspace, or Evaluator Trace. +Inspect only the requested Test Agent and Benchmark: the Agent State, public Statements, Scoreboard, and score-linked Test Traces or artifacts from the Baseline and this optimization, including rejected Candidates. -## State editing policy +Do not inspect Rubrics, Gold answers, private scoring conditions, Evaluator State, Workspace, or Trace, other Agents, or Project secrets. If private evaluation information enters the Optimizer context, restore the active Candidate and stop as contaminated. -Make the smallest complete edit supported by evidence and preserve unrelated instructions and files. - -- Behavioral, workflow, role, or domain guidance belongs in `agent_state/AGENTS.md` unless a relevant target-owned Skill already owns that reusable capability. -- Update a relevant target-owned `SKILL.md` when the behavior is a reusable capability shared across tasks. -- Create a narrowly named Skill only when the capability is reusable and no suitable Skill exists. Installing its directory is sufficient; do not register Skill metadata in AGENTS.md. -- Runtime limits belong in safe `system_config.yaml` fields. Do not edit `system_prompt` unless the user explicitly asks. -- Never modify a library-provided Skill such as `penguin-sdk` to carry target-specific behavior. - -Every edit must generalize beyond the observed run. Do not encode Case ids, exact expected outputs, Benchmark-specific constants, private criteria, or a guessed answer. Keep a recovered mapping, threshold, or formula only when repeated public evidence supports it as a durable rule; otherwise encode the reasoning and validation method. - -## Version and rollback discipline - -Before changing Agent State, read its canonical top-level `version` from `system_config.yaml`, defaulting to 1. The system owns Agent State snapshot archives and exposes them through Web export and import. Do not create, import, extract, or replace snapshot archives yourself. Require `/snapshots/v.tar.gz` to exist; if it is missing, stop and ask the user to export the current Agent State from Agent settings before continuing. - -For each file the edit will change, record its exact original bytes and whether it existed in a temporary directory outside `STATE`. Never include `.vault.toml`, an unrelated State file, or a snapshot archive. Write each candidate file through a temporary sibling, validate it, and rename it into place. Set the State version to `current + 1` exactly once before evaluation. - -If the candidate is rejected or a valid comparison cannot complete, restore only those recorded files, remove files created by the candidate, and verify that the prior version is active. If the active version or a candidate-owned file no longer matches the value written by this round, treat it as a concurrent mutation and stop without overwriting it. If rollback cannot be verified, stop and ask the user to restore the system snapshot through Web import. - -## One-shot feedback mode - -Use the user's feedback and relevant recent Trace when available. Turn that evidence into the smallest targeted change: - -- behavior, workflow, role, or domain guidance → update AGENTS.md or the relevant target-owned Skill; -- a missing reusable capability → install or update a narrowly scoped Skill; -- a runtime limit → adjust the relevant `system_config.yaml` field. - -Require the current-version system snapshot and record the exact originals of files being changed before editing. Increment the version once after the complete edit. If the edit cannot complete, roll back only those files. Report the evidence, changed State surface, new version, and reason. Do not claim measured improvement because this mode has no Benchmark comparison. - -## Benchmark optimization mode - -Require `benchmark_config.toml` with a positive integer `runs` and a complete baseline evaluation. The baseline Case set is frozen for optimization. Each reference or candidate evaluation must include exactly `runs` uniquely numbered runs for every frozen Case. - -Use the reference evaluation's `(provider, model_id)` pair unchanged so scores remain comparable. If the user wants a different Model, stop and ask for a new baseline series rather than comparing across Models. - -Before any candidate edit, read the canonical top-level State version and select a reference evaluation in the existing `(provider, model_id)` series. The selected reference is valid only when its `version` equals the active State version and its Cases and runs form the complete frozen Case × `runs` matrix. Never compare a candidate against an older-version, incomplete, or mixed-Model evaluation. - -If the active State has no such evaluation, measure it before optimizing: leave State unchanged, run the complete frozen matrix with the existing series' exact `(provider, model_id)` pair, and validate the results under the same rules used for a candidate. Retain the exact Scoreboard bytes and active State version before dispatch; append the completed no-edit reference through a temporary sibling, YAML validation, and atomic rename only if both remain unchanged. This appended evaluation becomes the reference. If any cell remains invalid after the allowed retry, provenance does not match, State or Scoreboard changes, or the atomic append cannot complete, stop before editing Agent State. - -You may inspect the complete target `agent_state/`, public Case Statements, the Scoreboard, and all Test traces referenced by the Scoreboard runs. Use only those explicit Case and Session ids. Never edit the Benchmark, Test traces, Project configuration, or another Agent. - -Scoreboard v2 uses this shape: - -```yaml -evaluations: - - time: "2026-07-17T00:00:00Z" - version: 2 - provider: deepseek - model_id: deepseek-v4-pro - summary_title: "Improved evidence validation" - summary: "Added a reusable validation step; all Cases improved without new instability." - score: 24 - cost: 0.04 - duration_ms: 60000 - cases: - - case: CASE-001-example - score: 24 - cost: 0.04 - duration_ms: 60000 - runs: - - score: 23 - cost: 0.03 - duration_ms: 58000 - session_id: session-1 - - score: 24 - cost: 0.04 - duration_ms: 60000 - session_id: session-2 - - score: 25 - cost: 0.05 - duration_ms: 62000 - session_id: session-3 -``` - -Case metrics are the means of valid `runs`; evaluation totals are the sums of Case means. Omit cost when any contributing run has unknown cost. `summary_title` and `summary` may describe public State changes, gains, regressions, and instability, but must not reveal private Rubric or Gold content, expected answers, or private scoring reasoning. +Modify only the Test Agent State and the versioned snapshot required to protect it. Do not change the frozen Benchmark, Test Traces, or Project configuration. The only Benchmark write is appending a complete accepted Candidate Evaluation to `scoreboard.yaml`. ## Optimization loop For each round: -1. Reconfirm that the reference version equals the active top-level State version and that it contains the complete frozen Case × `runs` matrix. Then analyze its aggregate, Case scores, repeated runs, and every score-linked Test Trace. Use repeated runs to separate stable failure from variation; never select a convenient Trace. -2. State one falsifiable behavioral hypothesis connecting public evidence to a minimal State change. If no credible hypothesis remains, stop. -3. Confirm the current-version system snapshot exists, record the exact originals of every candidate-owned file, make the candidate edit, and set `version` to `current + 1` once. -4. Retain the exact Scoreboard bytes and candidate version. Build the complete Case-run matrix before dispatch. -5. Start one child per cell with `run_subagent`, and omit `agent_id` so the child reuses the current Agent. Each prompt begins with the caller identity and the sentence: Use the `agent-evaluation` Skill. Then it contains one request: +1. **Establish the Reference.** Confirm that its complete Evaluation uses the frozen Case × Run matrix and evaluation runtime and that its version matches the current Agent State. +2. **Diagnose capability gaps.** Compare each Case's `runs[].score` on the fixed `0..100` scale; use the Evaluation's top-level average `score` only for whole-version comparison. Use public Statements, score-linked Test Traces, and prior accepted or rejected attempts to identify observable behaviors that general Agent State changes could improve. Use repeated Runs to distinguish stable behavior from variation. +3. **State a falsifiable hypothesis.** Choose the related gaps to address, connect them to a bounded Candidate, and state which observable decisions or artifacts should change and why. A change that only adds analysis steps without predicting a behavioral change is not a useful hypothesis. If the current diagnosis is exhausted, use the remaining public evidence and prior attempts to construct a different admissible Candidate. +4. **Create one Candidate from the Reference.** Apply the change and its Candidate version under the construction and rollback rules below. Do not carry rejected Candidate files into the next attempt. +5. **Check admissibility.** Confirm that the change is general, uses no private evaluation information, and modifies only permitted Test Agent State. +6. **Evaluate the Candidate.** Delegate the complete frozen Case × Run matrix in parallel under the evaluation rules below and assemble all returned cells. Do not modify the Candidate while any cell is in flight. +7. **Decide.** Accept the Candidate only when every cell is valid and its Evaluation's top-level average `score` is strictly higher than the Reference Evaluation's `score`. Otherwise restore the Reference. Record separately whether the predicted Case behavior changed; a higher Evaluation score accepts the Candidate even when the stated hypothesis was not supported. +8. **Persist and continue.** Immediately append and verify every accepted Candidate Evaluation before starting another round. An accepted Candidate becomes the next Reference. Use valid results from rejected Candidates only as evidence for a later hypothesis. Stop when the Reference reaches the desired target. Otherwise complete the requested number of valid Candidate rounds unless infrastructure, contamination, concurrent State changes, or the inability to construct any admissible Candidate creates a concrete blocker. At the round limit, retain the highest-scoring accepted Reference. - ```text - Caller agent: - Use the `agent-evaluation` Skill. Return only its terminal protocol YAML. - protocol_version: 1 - case_id: - run: <1_based_run_index> - expected_version: - test_agent_id: - benchmark_dir: - provider: - model_id: - ``` +A round counts only after one Candidate has a complete valid Evaluation. Corrected requests, validity repairs, and evaluation retries do not consume the round limit. A complete valid Evaluation of a rejected Candidate does count. - For N Cases and R runs, emit all N × R independent calls in the same parallel tool-call group before waiting. Continue an active child through `input_subagent`; never duplicate it. -6. Parse only each child's last terminal `protocol_version: 1` YAML mapping. Keep every identity-matched `status: ok` result. Retry invalid cells once, dispatching all retries together with identical State, Model, Benchmark, and run identities. Never retry a valid scored cell. If any retry remains invalid, reject the round and roll back the candidate files. -7. Compute Case means and evaluation sums. Retain every score, cost when complete, duration, and Test Session id. Re-read State version and exact Scoreboard bytes, and verify that every candidate-owned file still matches the value written by this round. If State changed concurrently, stop without overwriting it. If only the Scoreboard changed, reject the round and roll back the candidate files. -8. Accept only a score strictly higher than the comparable reference evaluation. On improvement, keep the candidate State and append one evaluation through a temporary sibling, YAML validation, and atomic rename. On an equal or lower score, roll back the candidate files and append nothing. +## Build and roll back a Candidate -Each accepted round becomes the next reference. Stop when the user's target or round limit is met, no credible evidence-backed hypothesis remains, or infrastructure prevents another valid comparison. Do not mutate State as random search. +Create one Candidate per round from the current Reference. Put behavioral guidance in `AGENTS.md`, reusable target-owned capabilities in a focused Skill, and runtime limits in safe `system_config.yaml` fields. Do not edit `system_prompt` unless requested, modify library-provided Skills for target-specific behavior, or change `model.thinking_level`; the Reference Scoreboard fixes the evaluation thinking level. -At the end, report the accepted score curve, State versions and changes, rejected hypotheses and rollbacks, Test Session ids, stop reason, and limitations. Distinguish the active tested State from any unscored State; never claim a Scoreboard score applies to a later untested edit. +Candidate version numbers only increase. Start with `Reference version + 1` and never reuse a rejected version. Before changing the Agent State, save the original contents and record any files the Candidate creates. + +Before changing each Reference State, ensure `/snapshots/v.tar.gz` exists. Reuse it when present. Otherwise create it yourself before editing by atomically archiving `agent_state/` while excluding `.vault.toml`; validate the archived version and never overwrite an existing same-version snapshot. If snapshot creation fails, stop before changing Agent State and report the failure. + +Keep the exact original-file record for fast in-round rollback. + +If the Candidate is rejected or cannot be evaluated, restore the Reference files and version, remove files created by the Candidate, and verify the restoration. If another process changes the Agent State, stop without overwriting it. + +## Delegate evaluation + +For each Case × Run cell, call `run_subagent` with: + +```text +Use the `agent-evaluation` Skill. Run the specified Test Agent on the specified Case exactly once, then score that single execution. +protocol_version: 1 +case_id: +run: <1_based_run_index> +expected_version: +test_agent_id: +benchmark_id: +provider: +model_id: +``` + +Inspect the complete streamed and final worker response. Before reading `status`, `score`, or any other protocol field, verify that the worker-authored text is exactly one plain protocol YAML document. Narration, headings, code fences, summaries, or scoring details are not valid protocol. Ask the same Evaluator to resend only the clean YAML from its existing result; do not rerun the Test Agent for a formatting repair and do not extract YAML from the invalid response yourself. Transport metadata added by `run_subagent` is not worker-authored text. If private evaluation information appears, follow the contamination rule above. + +For every scored result, require its actual `provider`, `model_id`, and `thinking_level` to equal the Reference runtime. A mismatch invalidates the Candidate matrix and stops optimization; never compare or record scores produced under a different runtime. + +Correct and resend an `invalid_request`. Stop on `version_changed` or `benchmark_invalid`. + +For `evaluation_failed`, keep the same Candidate and incomplete matrix. Ask the same Evaluator to diagnose and repair the failed cell, then rerun only that cell when evidence proves the Test Agent did not start. Every retry must apply a new, specific repair; never repeat an unchanged request or launch, and do not impose a numeric retry limit while distinct safe repairs remain. Do not inspect private Evaluator State or abandon the Candidate to design the next version. Stop when no new safe repair remains, external configuration is required, or it is unclear whether the Test Agent started. + +## Record and report + +Append each complete accepted Candidate Evaluation to `scoreboard.yaml` immediately after acceptance and verify the stored version, score, matrix, and Session ids before continuing. Obtain the current UTC timestamp from the environment, for example with `date -u +"%Y-%m-%dT%H:%M:%SZ"`, rather than inferring UTC from a displayed local time. Use the same field names as the Baseline: + +```yaml +- time: + version: + provider: + model_id: + thinking_level: + summary_title: + summary: + score: + cost: + duration_ms: + cases: + - case: + score: + cost: + duration_ms: + runs: + - score: + cost: + duration_ms: + session_id: +``` + +Every Run and Case score is on the fixed `0..100` scale. Do not write `max_score`. Calculate and write every Case and Evaluation average directly in the Scoreboard: ignore `null` values when averaging cost and write `null` only when all contributing costs are unknown; round `score` averages to two decimal places, `cost` averages to six decimal places, and `duration_ms` averages to the nearest integer. These stored values are authoritative—do not add a server, frontend, script, or consistency check that recomputes or validates them. Do not add an `aggregate` object or use `case_id`, `mean_score`, `mean_cost`, or `mean_duration_ms`. Do not record rejected Candidates in the Scoreboard. + +Report the Baseline and every fully evaluated Candidate with its score, version, change, decision, and Test Session ids. For each Candidate, distinguish the acceptance decision from whether its stated hypothesis was supported by the predicted Case behavior. Include the final retained version, stop reason, and known limitations. Never report a score for an Agent State that was not evaluated. diff --git a/packages/skills/skills/benchmark-design/SKILL.md b/packages/skills/skills/benchmark-design/SKILL.md index d06e0f5..fcc267e 100644 --- a/packages/skills/skills/benchmark-design/SKILL.md +++ b/packages/skills/skills/benchmark-design/SKILL.md @@ -1,40 +1,57 @@ --- name: benchmark-design -description: Design and calibrate a multi-Case capability Benchmark with repeated independent evaluations and a traceable baseline. +description: Design and calibrate a multi-Case capability Benchmark and establish a traceable Formal Baseline. short_description: Design and calibrate an Agent capability Benchmark. short_description_zh: 设计并校准 Agent 能力评测 Benchmark。 -version: 4 -updated: 2026-07-25T00:00:00Z +version: 5 +updated: 2026-07-29T17:20:58Z --- # Benchmark Design -Create and calibrate a multi-Case Benchmark that discovers a specified Test Agent's capability boundary. Own the public Statements, private Rubrics, Case set, configuration, and final baseline. Do not modify the Test Agent State, launch the Test Agent directly, or score a Case yourself. +Build a multi-Case Benchmark for one Test Agent, calibrate its difficulty, and record a complete Formal Baseline. + +This Skill changes the Benchmark, never the Test Agent. It does not run or score the Test Agent. Delegate every evaluation with `run_subagent`, and tell each worker to use `agent-evaluation`. Stop after the Baseline; do not begin optimization. ## Before you start -Require a Test Agent and the capability to measure. If either is missing, ask the user. A Benchmark run also requires a top-level Session with `run_subagent` and the current Agent must have `agent-evaluation` installed. If the Skill is missing or this Session is already a subagent, stop and ask the user to install the Skill or start a top-level Session. Do not begin a partial Benchmark. +If the request does not identify a Test Agent, target capability, desired baseline score, and Pilot iteration limit, ask for the missing inputs. When they are already supplied, proceed without asking the user to restate them. Treat the current Agent as the **Builder**. A user-specified evaluation `(provider, model_id)` takes priority; otherwise inherit the current Builder Session's complete `Provider` and `Model ID` from the Environment. Never use a Project default as an implicit evaluation runtime. -## Boundaries +## Workflow -Access only the explicit Test Agent and Benchmark paths. Do not inspect another Agent, Project configuration files, Agent Evaluator State, Evaluator Workspace, or Evaluator Trace. Consume only each Evaluator's terminal protocol response and the returned Test Session id. +- A **Pilot** is a provisional evaluation used to improve the Benchmark. Its results never enter the Scoreboard. +- **Freeze** means the Benchmark revision and evaluation settings stop changing. +- A **Formal Baseline** is the accepted result of a fresh, complete Case × Run evaluation of the frozen Benchmark on one unchanged Agent State version. -Use the Environment's App Data Dir and the explicit Test Agent id: +Follow this order: + +1. Validate the Test Agent, target capability, resolved evaluation Runtime, and evaluation access. +2. Write a Capability Contract that defines the observable process to measure, common weaker behavior, and the general Agent State improvement the Benchmark should train. +3. Plan the complete initial Case set and point allocation. For each Case, privately state the intended behavior, a plausible shortcut for a strong Test Agent, and how the Case distinguishes them. Write and leak-check the complete initial Benchmark. +4. Complete one valid evaluation for every planned Case. Together these results form Pilot iteration 1; finish this complete set before refining any Case. +5. For later Pilot iterations, use scores and Traces to reconstruct how the Test Agent solved each Case. A single iteration may refine multiple Cases or difficulty dimensions; rerun every affected Case. +6. Freeze the first valid Pilot revision that meets the desired baseline score. If none does within the requested valid-iteration limit, restore and freeze the lowest-scoring valid Pilot revision. +7. Run a fresh, complete Case × Run matrix and save it as the Formal Baseline when every cell is valid, the Agent State version remains unchanged, and no known design defect remains. The Formal score does not determine validity. + +## Setup and access + +Require a Test Agent id, target capability, desired baseline score on the fixed `0..100` scale, and a positive Pilot iteration limit. Derive a short semantic Benchmark id if needed. Resolve `(provider, model_id)` once before the first Pilot: use a user-supplied complete pair when present, otherwise inherit the current Builder Session's `Provider` and `Model ID` from the Environment. Reject a half pair or an unavailable inherited value. Read `thinking_level` from the Test Agent's `model.thinking_level` in `agent_state/system_config.yaml`, using the normal Agent-config default `medium` only when that field is absent. Do not read `thinking_level` from a Trace and do not inspect Project configuration. + +The current Session must provide `run_subagent`, and the current Agent must have `agent-evaluation` installed. If either is missing, stop and explain what is needed. + +Use the Environment's `App Data Dir` and the explicit Test Agent id: ```text -PROJECT_DIR = -PROJECT_ID = -PENGUIN_HOME = TEST_AGENT_DIR = /agents/ BENCHMARK_DIR = /agents//benchmarks/ SCOREBOARD = /scoreboard.yaml ``` -Derive a semantic Benchmark id when the user does not supply one. Require `agent_state/system_config.yaml`; its top-level `version` is the canonical State version and defaults to 1 when absent. +Access only the specified Test Agent and Benchmark: the Agent State, complete Benchmark, and Test Traces or artifacts from valid evaluations. Do not access other Agents, Project secrets, or Evaluator State, Workspace, or Trace. -## Benchmark contract +Read the Agent State version from the top-level `version` in `agent_state/system_config.yaml`; use 1 only when it is absent. -Use this structure: +## Build the Benchmark ```text / @@ -42,115 +59,143 @@ Use this structure: ├── scoreboard.yaml └── CASE--/ ├── statement/ - │ └── README.md + │ ├── README.md + │ └── └── rubric/ └── README.md ``` -Both README files are required; either directory may contain supporting files. `statement/` is the complete public task and evidence. `rubric/` is private scoring material. Never mention private criteria or paths in the Statement. +Each Case contains: -Create `benchmark_config.toml` first. The Model is deliberately not stored here; each evaluation records the actual `(provider, model_id)` pair. +- `statement/`, which is public to the Test Agent and defines the objective, available materials, and required artifact. +- `rubric/`, which is private and defines observable scoring items, points, and Gold answers. -```toml -title = "" -description = "" -runs = 3 -``` +Both directories require a `README.md` and may contain supporting files. Do not put Gold answers for evaluated instances, hidden mappings, or private scoring conditions in `statement/`. -Use `runs = 3` unless the user explicitly requests another positive integer. Initialize the Scoreboard with: +Create `benchmark_config.toml` with `title`, `description`, and `runs = 3`. Use another positive Run count only when requested. Initialize `scoreboard.yaml` with `evaluations: []`. -```yaml -evaluations: [] -``` +Pass the resolved `(provider, model_id)` explicitly in every Pilot and Formal Evaluator request, starting with the first cell. Freeze that pair and the Test Agent's configured `thinking_level` for the complete Benchmark workflow. Every scored Evaluator result must report the requested pair and the same configured thinking level. A mismatch invalidates the matrix. -Every accepted evaluation follows scoreboard v2: +Before planning Cases, state the Capability Contract: -```yaml -evaluations: - - time: "2026-07-17T00:00:00Z" - version: 1 - provider: deepseek - model_id: deepseek-v4-pro - summary_title: "Calibrated baseline" - summary: "Abbreviated baseline schema for one Case with three independent runs." - score: 18 - cost: 0.04 - duration_ms: 60000 - cases: - - case: CASE-001-example - score: 18 - cost: 0.04 - duration_ms: 60000 - runs: - - score: 17 - cost: 0.03 - duration_ms: 58000 - session_id: session-1 - - score: 18 - cost: 0.04 - duration_ms: 60000 - session_id: session-2 - - score: 19 - cost: 0.05 - duration_ms: 62000 - session_id: session-3 -``` +- the public evidence available to the Test Agent; +- the observable decisions, intermediate artifacts, and checks the capability requires; +- the weaker behaviors or shortcuts the Benchmark should distinguish; and +- the reusable Agent State behavior that could improve the measured capability. -Case `score`, `cost`, and `duration_ms` are the means of their valid `runs`. Evaluation totals are the sums of the Case means. Omit a Case or evaluation `cost` when any contributing run has unknown cost; never treat unknown as zero. Rubric maxima across the complete Case set should total 100 points so the evaluation score remains interpretable on a 0–100 scale. +Before writing each Case, privately record the required behavior, a plausible shortcut for a strong Test Agent, the chosen difficulty, the different scored decision or artifact each behavior should produce, and why the distinction measures the target capability. Design the Case so the measured capability affects the score. Do not optimize the Statement to help the Test Agent succeed or copy this design rationale into it. -Builder writes one final baseline for the current Benchmark definition. A material change to a Statement, Rubric, Case set, or `runs` invalidates prior results: clear `evaluations`, recalibrate, and write a new baseline. Rejected candidates and provisional matrices remain only in Builder Trace. Evaluator never writes the Scoreboard. +The Statement presents the task, not the Benchmark's teaching or design intent. It describes the objective, available materials, option meanings, output format, and necessary constraints. It must not prescribe the reasoning sequence, identify decisive evidence, name the shortcut, or reveal private scoring preferences. When an auditable artifact is needed, request concise supporting evidence without prescribing how to obtain it. -The public `summary_title` and `summary` may describe the tested State, Case-level score patterns, and instability. They must not reveal Rubric or Gold content, expected answers, private scoring reasoning, mappings, thresholds, formulas, or rules. +Keep the evaluation contract well-defined, but do not require the public Statement to uniquely determine the Gold. Public information may be incomplete or conflicting, and the Rubric may encode a private decision standard or preference. Fix that private standard before evaluating the revision and never change its Gold after seeing the evaluated answer. The standard must remain tied to the target capability: it should express a stable reusable policy, priority, inference boundary, or other behavior that a better Agent State could apply across instances. Do not use a capability-irrelevant random hidden mapping merely to lower the score, and do not disclose every decisive premise or priority merely to make the public task complete. -## Design and calibrate +The first complete revision is an exploratory probe. Use its Pilot to learn how the Test Agent interprets the tasks, forms candidate rules, and uses shortcuts; refine the Benchmark before treating it as calibrated. A later revision may intentionally add information gaps, conflicts, private preferences, or other capability-relevant distinctions in response to an earlier Trace, provided the next revision's Rubric is fixed before dispatch. -Before writing Cases, define the observable difference between an Agent that has the requested capability and one that does not. Each Case must make that capability causally necessary, not merely share its topic. +Every Case Rubric has a fixed maximum of 100 points, with observable scoring items and meaningful partial credit. Allocate most points within each Case to decisions or concise artifacts on which the intended behavior and plausible shortcut differ. Keep generic format compliance, evidence enumeration, and analysis completeness from creating a high score floor unless those are themselves the target capability. Allocate points from capability coverage before the first Pilot. Do not change scoring items solely to satisfy the desired score; when a redesign changes coverage, re-plan that Case's 100-point allocation before evaluating the revised Case set. When final choices do not distinguish the intended behavior from a shortcut, score a concise auditable artifact, but define only its required content or format—not the method used to produce it. -Run this counterfactual before accepting a Case: could a competent executor without the target capability complete it by mechanically following the Statement? If yes, reject or redesign it. A self-contained Statement specifies the task, available evidence, and required artifact without disclosing the reasoning, mapping, or rule the capability is supposed to recover. The evidence must still make the answer inferable. +Before the first dispatch of every new or changed Case revision, run a consistency review: -Build several independent, realistic end-to-end Cases with distinct capability-relevant failure modes. Do not manufacture low scores through missing essential evidence, trivia, formatting traps, excessive workload, or unstable infrastructure. +- Confirm that the current Statement is internally coherent. Intentional conflicts must be presented as conflicts between sources, rules, or positions rather than as contradictory claims by the Benchmark itself. +- Confirm that the current Rubric is consistent with the current Statement and fixed private standard. It must be self-contained and must not refer to an earlier revision or missing context. +- Confirm that every scoring item applies to the Case's actual requested output and relies only on premises that are defined, provided, or explicitly private under the fixed standard. -Freeze every Rubric before evaluation. Use atomic observable conditions, exact points, reasonable equivalence rules, and meaningful partial credit. Test the Rubric mentally against full, partial, missing, malformed, wrong-type, and extra output. Never execute Test Agent-produced code while scoring. +This review does not require the public Statement to contain enough information to reproduce the private standard or uniquely derive every Gold answer. Unchanged Cases do not need another review during that iteration. Keep this review in Builder analysis and Trace; fix defects in the Case rather than creating a separate audit artifact. -Use valid evidence to calibrate toward a user-supplied target; otherwise aim near 60/100. Treat a near-ceiling candidate as uncalibrated when a credible structural refinement remains. Audit high scores for shortcuts or leakage and low scores for ambiguity, missing evidence, unrelated difficulty, or a defective Rubric. More items, steps, or workload alone are not structural refinement. +Also compare all public files with the private Rubric. Confirm that no public file reveals Gold answers, private scoring conditions, or hints that identify the intended solution. This is the leak check. -## Select the evaluation Model +## Delegate evaluation -Use a user-specified `(provider, model_id)` pair when supplied. Otherwise resolve the Project default with the supported CLI, never by reading the hidden Project configuration: - -```bash -penguin config model list --project-id "" --root "" -``` - -The row marked `*` is the default. Keep the same pair through all candidate matrices for this calibration. The pair selects the Test Agent's CLI Session, not the Builder or Evaluator runtime model. - -## Run the Case-run matrix - -Read and retain the exact State version, Scoreboard bytes, configured positive `runs`, selected Model pair, and complete valid Case set. Build every unique Case-run cell before dispatch. - -Start one child per cell with `run_subagent`, and omit `agent_id` so the child reuses the current Agent. Each prompt must begin with the caller identity and the sentence: Use the `agent-evaluation` Skill. Then provide exactly one request: +For each Case × Run cell, call `run_subagent` with the request below. Dispatch independent cells in parallel up to available concurrency. ```text -Caller agent: -Use the `agent-evaluation` Skill. Return only its terminal protocol YAML. +Use the `agent-evaluation` Skill. Run the specified Test Agent on the specified Case exactly once, then score that single execution. protocol_version: 1 case_id: run: <1_based_run_index> -expected_version: +expected_version: test_agent_id: -benchmark_dir: +benchmark_id: provider: -model_id: +model_id: ``` -For N Cases and R runs, emit all N × R independent `run_subagent` calls in the same parallel tool-call group before waiting for results. Continue an active child through `input_subagent`; never duplicate it. +Inspect the complete streamed and final worker response. Before reading `status`, `score`, or any other protocol field, verify that the worker-authored text is exactly one plain protocol YAML document. Narration, headings, code fences, summaries, or scoring details are not valid protocol. Ask the same Evaluator to resend only the clean YAML from its existing result; do not rerun the Test Agent for a formatting repair and do not extract YAML from the invalid response yourself. Transport metadata added by `run_subagent` is not worker-authored text. A wrong or missing Test Agent artifact is a valid scored result and must not be retried. -Parse only the last terminal `protocol_version: 1` YAML mapping. Keep every identity-matched `status: ok` result. Retry only invalid cells once, with all retry cells dispatched together and the same State version, Model, Benchmark, and run identities. Never retry a valid scored cell. If any retry remains invalid, abandon the matrix. +For every scored result, require non-empty `provider`, `model_id`, and `thinking_level`. Require the model pair to equal the explicitly resolved pair and the thinking level to equal the Test Agent configuration read before dispatch. Reject a Pilot or Formal matrix whose cells report mixed or mismatched runtimes. The Evaluator verifies provider/model from the root Trace and reports thinking from the unchanged Target Agent configuration; it does not require Trace metadata for thinking. -For a complete matrix, calculate Case means and evaluation sums using scoreboard v2. Retain every run's score, cost when known, duration, and Test Session id. Re-read State version and exact Scoreboard bytes; abandon the result if either changed. +Correct and resend an `invalid_request`. For `benchmark_invalid`, repair and rerun the affected Case during Pilot; during Formal, abandon the matrix and return to Pilot. For `version_changed`, discard the matrix and restart after the Agent version is stable. -Use each returned Test Session id to inspect the exact Test Trace and artifact. Analyze all repeated runs; disagreement is capability instability, not permission to select a convenient result. Accept a candidate only when the capability caused the scored difference, evidence was sufficient, the Rubric was sound, and useful headroom remains. +For `evaluation_failed`, keep the same Benchmark revision and cell. Diagnose the failure and retry only when evidence proves the Test Agent did not start and the retry applies a new, specific repair. Do not set a numeric retry limit or repeat an unchanged launch. Stop when no new safe repair remains, external configuration is required, or it is unclear whether the Test Agent started. Never treat an evaluation failure as score zero. -When calibration stops after a complete valid matrix, write the final baseline to a temporary sibling, parse it as YAML, then atomically rename it over `scoreboard.yaml`. Include the sorted Case set and sorted runs, a real UTC ISO-8601 time, the tested version and Model pair, and a privacy-safe summary. A near-ceiling final score is still recorded, but if the refinement budget ends without an acceptable candidate, report `calibration_failed` rather than calling the Benchmark ready. Leave `evaluations` empty only when no complete valid matrix exists. +## Refine the Benchmark -Report the Benchmark path, aggregate and Case scores, Test Session ids, refinements, stop reason, and limitations. +Treat the first draft as a hypothesis. The first valid result from every planned Case together forms Pilot iteration 1. A later iteration starts after a difficulty refinement and completes when every affected Case has a valid new result. Request corrections, validity repairs, and evaluation reruns stay in the current iteration and do not consume the requested iteration budget. Use the recorded Agent State version and fixed evaluation runtime. + +Keep Pilot results out of the Scoreboard. During calibration, retain only one temporary restorable copy: the lowest-scoring complete valid revision seen so far. Store it outside `benchmarks/`, replace it only when a lower valid revision completes, and never retain invalid revisions. + +Use the Pilot to find the current Test Agent's capability boundary. + +Before editing, distinguish a validity repair from a difficulty refinement. A validity repair fixes an unusable task or scoring contract and stays in the current Pilot iteration. A difficulty refinement changes what the valid Benchmark measures and completes the next iteration after every affected Case has a valid result. + +Before editing, estimate how much of the score the planned refinements can affect. If the range is too small to materially approach the desired score, revise more affected Cases, use more than one difficulty dimension, or replace low-signal Cases. + +Prefer refinements that create one or more scored separating decisions. A refinement may change the public task or evidence, introduce or preserve a reasonable information gap or conflict, or apply a fixed private standard. Adding another explicit rule, exception, source, or checklist is not a difficulty increase when the observed strategy can still follow it to the Gold. A Rubric-only refinement is allowed but not preferred when the public task already contains the relevant information, the current Rubric fails to distinguish merely mentioning it from handling it correctly, and the Builder can explain which reusable capability the new scoring distinction measures. Do not add points merely because the previous Test Agent omitted a phrase. Fix the revised Rubric before dispatch and treat it as a changed Case revision. + +For each refinement iteration: + +1. **Observed strategy.** Reconstruct the Test Agent's actual solution method from its score, artifact, and Trace. +2. **Missing behavior.** Identify the general behavior that the observed strategy skipped or simplified. Repair missing evidence, arbitrary mappings, ambiguity, or scoring defects before increasing difficulty. +3. **Separating prediction.** Before dispatch, predict the decision or artifact the observed strategy will produce, the different result the desired behavior will produce, and the score range affected. If both behaviors are expected to reach the same scored result, choose another refinement. +4. Update any number of diagnosed Cases or difficulty dimensions, run the consistency review and leak check for each changed revision, and rerun every affected Case. + +Reuse a Pilot result only when the Case revision, scoring, Agent State version, and evaluation runtime are unchanged. + +An information gap or supported alternative is not automatically a design defect. Treat it as a defect only when the task or fixed private standard is incoherent, changes after evaluation, leaks the answer, or no reusable Agent behavior could plausibly improve the score. + +More rows, fields, distractors, files, near-duplicate examples, or explicit rule layers do not increase difficulty when the observed strategy still solves the Case. Base refinements on observed behavior and fix the Gold before each evaluation. + +Freeze immediately when a complete valid Pilot iteration meets the desired baseline score and no known design defect remains. Do not run another difficulty refinement merely to create more score margin. Otherwise continue through the requested valid-iteration limit. If the desired score is still unmet, restore the temporary lowest-scoring valid revision and proceed to Freeze. Report `calibration_failed` only when no valid Pilot revision can be produced or evaluation failures prevent a valid selection; missing the desired score alone is not a failure. + +## Freeze and run the Formal Baseline + +After selecting the Pilot revision, restore that exact revision if needed. Run a complete consistency review and final leak check across every Case, fix any defect, then freeze the Benchmark and record the current Agent State version. Run a fresh, complete Case × Run matrix and never reuse a Pilot result. Once the first Formal cell is dispatched, do not change the Benchmark. + +Accept the matrix when every cell is valid, every cell reports the frozen evaluation runtime, the Agent State version remains unchanged, the private scoring standard remained fixed, and every score loss reflects the Capability Contract. Record the Formal Baseline even when its score does not meet the desired baseline score. + +If Formal reveals a design defect, abandon the matrix and repair the frozen candidate revision before freezing and rerunning the complete matrix. Report `calibration_failed` only when no valid revision remains or evaluation failures prevent a complete Formal matrix. Never record a partial, abandoned, or invalid Formal matrix. + +## Record and finish + +After validation, obtain the current UTC timestamp from the environment, for example with `date -u +"%Y-%m-%dT%H:%M:%SZ"`, rather than inferring UTC from a displayed local time. Append only the accepted Formal Baseline to `scoreboard.yaml` using exactly this structure: + +```yaml +evaluations: + - time: + version: + provider: + model_id: + thinking_level: + summary_title: + summary: + score: + cost: + duration_ms: + cases: + - case: + score: + cost: + duration_ms: + runs: + - score: + cost: + duration_ms: + session_id: +``` + +Every Run and Case score is on the fixed `0..100` scale. Do not write `max_score`. Calculate and write every Case and Evaluation average directly in the Scoreboard: ignore `null` values when averaging cost and write `null` only when all contributing costs are unknown; round `score` averages to two decimal places, `cost` averages to six decimal places, and `duration_ms` averages to the nearest integer. These stored values are authoritative—do not add a server, frontend, script, or consistency check that recomputes or validates them. Do not add an `aggregate` object or use `case_id`, `mean_score`, `mean_cost`, or `mean_duration_ms`. + +Report the Benchmark path, configuration, Agent State version, Evaluation average and Case Run scores, Test Session ids, and known limitations. Include one compact row per Pilot iteration with its score, diagnosed capability gap, difficulty adjustment, and freeze or stop decision. + +After the accepted Formal Baseline is recorded, delete the temporary lowest-revision copy and other Builder calibration scaffolding. Keep the frozen Benchmark, Scoreboard, evaluation Workspaces, and score-linked Traces. + +Do not reveal Rubrics, Gold answers, latent rules, per-item scores, or private scoring information. Stop after reporting the Baseline; do not modify the Test Agent or begin optimization. diff --git a/packages/skills/test/skills.test.ts b/packages/skills/test/skills.test.ts index 7d79d36..cd540c6 100644 --- a/packages/skills/test/skills.test.ts +++ b/packages/skills/test/skills.test.ts @@ -182,6 +182,169 @@ describe("librarySkill", () => { expect(librarySkill("no-such-skill")).toBeUndefined(); }); + it("benchmark-design selects a valid Pilot revision before the Formal Baseline", () => { + const content = librarySkill("benchmark-design")?.content; + expect(content).toBeDefined(); + + const pilotIndex = content!.indexOf("## Refine the Benchmark"); + const baselineIndex = content!.indexOf("## Freeze and run the Formal Baseline"); + const normalizedContent = content!.replace(/\s+/g, " "); + + expect(pilotIndex).toBeGreaterThan(-1); + expect(baselineIndex).toBeGreaterThan(pilotIndex); + expect(normalizedContent).toContain("Keep Pilot results out of the Scoreboard"); + expect(normalizedContent).toContain("requested valid-iteration limit"); + expect(normalizedContent).toContain("lowest-scoring valid Pilot revision"); + expect(normalizedContent).toContain("retain only one temporary restorable copy"); + expect(normalizedContent).toContain("Store it outside `benchmarks/`"); + expect(normalizedContent).toContain("delete the temporary lowest-revision copy"); + expect(normalizedContent).toContain("A single iteration may refine multiple Cases"); + expect(normalizedContent).toContain( + "do not require the public Statement to uniquely determine the Gold", + ); + expect(normalizedContent).toContain("stable reusable policy, priority, inference boundary"); + expect(normalizedContent).toContain( + "do not disclose every decisive premise or priority merely to make the public task complete", + ); + expect(normalizedContent).toContain( + "Allocate most points within each Case to decisions or concise artifacts", + ); + expect(normalizedContent).toContain( + "Keep generic format compliance, evidence enumeration, and analysis completeness from creating a high score floor", + ); + expect(normalizedContent).toContain( + "Before the first dispatch of every new or changed Case revision", + ); + expect(normalizedContent).toContain("current Statement is internally coherent"); + expect(normalizedContent).toContain("current Rubric is consistent with the current Statement"); + expect(normalizedContent).toContain("must not refer to an earlier revision or missing context"); + expect(normalizedContent).toContain( + "every scoring item applies to the Case's actual requested output", + ); + expect(normalizedContent).toContain( + "does not require the public Statement to contain enough information", + ); + expect(normalizedContent).toContain( + "Prefer refinements that create one or more scored separating decisions", + ); + expect(normalizedContent).toContain( + "Adding another explicit rule, exception, source, or checklist is not a difficulty increase", + ); + expect(normalizedContent).toContain("Separating prediction"); + expect(normalizedContent).toContain( + "If both behaviors are expected to reach the same scored result, choose another refinement", + ); + expect(normalizedContent).toContain("A Rubric-only refinement is allowed but not preferred"); + expect(normalizedContent).toContain( + "Do not add points merely because the previous Test Agent omitted a phrase", + ); + expect(normalizedContent).toContain( + "Run a complete consistency review and final leak check across every Case", + ); + expect(normalizedContent).toContain( + "Do not run another difficulty refinement merely to create more score margin", + ); + expect(normalizedContent).toContain("missing the desired score alone is not a failure"); + expect(normalizedContent).toContain("never reuse a Pilot result"); + expect(normalizedContent).toContain( + "Record the Formal Baseline even when its score does not meet the desired baseline score", + ); + }); + + it("evaluation is the only subagent leaf and optimization has a bounded valid-round loop", () => { + const creation = librarySkill("agent-creation")!.content.replace(/\s+/g, " "); + const evaluation = librarySkill("agent-evaluation")!.content.replace(/\s+/g, " "); + const benchmarkDesignRaw = librarySkill("benchmark-design")!.content; + const benchmarkDesign = benchmarkDesignRaw.replace(/\s+/g, " "); + const optimizationRaw = librarySkill("agent-optimization")!.content; + const optimization = optimizationRaw.replace(/\s+/g, " "); + + expect(creation).toContain("inherit the current Builder Session's `Provider` and `Model ID`"); + expect(creation).toContain( + "read `model.thinking_level` from the Builder's own `agent_state/system_config.yaml`", + ); + expect(creation).toContain("Write the resolved `thinking_level` into a brand-new target Agent"); + expect(creation).toContain("never add either field to `system_config.yaml`"); + expect(evaluation).toContain("request from a `run_subagent` caller"); + expect(evaluation).toContain("Use the Penguin CLI only to launch the specified Test Agent"); + expect(evaluation).toContain('PROJECT_ID="$(basename "$PROJECT_DIR")"'); + expect(evaluation).toContain('PENGUIN_HOME="$(dirname "$PROJECT_DIR")"'); + expect(evaluation).toContain("export PENGUIN_HOME"); + expect(evaluation).toContain('--project-id "$PROJECT_ID"'); + expect(evaluation).toContain("Perform these as separate shell statements in this order"); + expect(evaluation).toContain("Never compress the assignments onto one command line"); + expect(evaluation).toContain("Do not impose a numeric retry limit"); + expect(evaluation).toContain("Never browse, query a pricing service"); + expect(evaluation).toContain("resend only the clean protocol YAML"); + expect(evaluation).toContain('--workspace ""'); + expect(evaluation).toContain("resolve it to an absolute canonical path"); + expect(evaluation).toContain("`provider` and `model_id` must both be non-empty"); + expect(evaluation).toContain("Never omit either model flag"); + expect(evaluation).toContain( + "Read and snapshot `model.thinking_level` from this Target Agent config", + ); + expect(evaluation).toContain("not Trace metadata—for `thinking_level`"); + expect(evaluation).toContain("outside `0..100`"); + expect(evaluation).not.toContain('PENGUIN_HOME="$(dirname "$PROJECT_DIR")" penguin run'); + expect(benchmarkDesign).toContain( + "otherwise inherit the current Builder Session's complete `Provider` and `Model ID`", + ); + expect(benchmarkDesign).toContain( + "Read `thinking_level` from the Test Agent's `model.thinking_level`", + ); + expect(benchmarkDesign).toContain( + "Pass the resolved `(provider, model_id)` explicitly in every Pilot and Formal Evaluator request", + ); + expect(benchmarkDesign).toContain("Every Case Rubric has a fixed maximum of 100 points"); + expect(benchmarkDesign).toContain("average of the Case scores"); + expect(benchmarkDesign).toContain("ignore `null` values when averaging cost"); + expect(benchmarkDesign).toContain("stored values are authoritative"); + expect(benchmarkDesign).toContain("obtain the current UTC timestamp from the environment"); + expect(benchmarkDesignRaw).not.toMatch(/\n\s+max_score:/); + expect(benchmarkDesign).toContain( + "Before reading `status`, `score`, or any other protocol field", + ); + expect(benchmarkDesign).toContain("Ask the same Evaluator to resend only the clean YAML"); + expect(optimization).toContain("Delegate every evaluation to an `agent-evaluation` subagent"); + expect(optimization).toContain("Before reading `status`, `score`, or any other protocol field"); + expect(optimization).toContain("Ask the same Evaluator to resend only the clean YAML"); + expect(optimization).toContain( + "A round counts only after one Candidate has a complete valid Evaluation", + ); + expect(optimization).toContain("keep the same Candidate and incomplete matrix"); + expect(optimization).toContain( + "Do not inspect private Evaluator State or abandon the Candidate", + ); + expect(optimization).toContain( + "Immediately append and verify every accepted Candidate Evaluation", + ); + expect(optimization).toContain( + "distinguish the acceptance decision from whether its stated hypothesis was supported", + ); + expect(optimization).toContain( + "At the round limit, retain the highest-scoring accepted Reference", + ); + expect(optimization).toContain("Read the evaluation `(provider, model_id, thinking_level)`"); + expect(optimization).toContain( + "require its actual `provider`, `model_id`, and `thinking_level` to equal the Reference runtime", + ); + expect(optimization).toContain("Do not edit `system_prompt` unless requested"); + expect(optimization).toContain("change `model.thinking_level`"); + expect(optimization).toContain( + "ensure `/snapshots/v.tar.gz` exists", + ); + expect(optimization).toContain("Reuse it when present"); + expect(optimization).toContain("create it yourself before editing"); + expect(optimization).toContain("excluding `.vault.toml`"); + expect(optimization).toContain("never overwrite an existing same-version snapshot"); + expect(optimization).not.toContain("stop and ask the user to export"); + expect(optimization).toContain("fixed `0..100` scale"); + expect(optimization).toContain("average of the Case scores"); + expect(optimization).toContain("stored values are authoritative"); + expect(optimization).toContain("Obtain the current UTC timestamp from the environment"); + expect(optimizationRaw).not.toMatch(/\n\s+max_score:/); + }); + it("rejects illegal-character names (path traversal guard) and never hits the filesystem", () => { for (const name of ["../penguin-sdk", "..", "penguin-sdk/SKILL.md", "a/../b", ".", ""]) { expect(librarySkill(name), name).toBeUndefined(); diff --git a/packages/web/src/api/endpoints.ts b/packages/web/src/api/endpoints.ts index 3d94aeb..18b7ef0 100644 --- a/packages/web/src/api/endpoints.ts +++ b/packages/web/src/api/endpoints.ts @@ -21,6 +21,7 @@ import type { ApprovalDecisionRequest, AuthLoginRequest, AuthResponse, + BenchmarkCasesResponse, BenchmarksResponse, DirListResponse, FilesStatRequest, @@ -522,6 +523,44 @@ export const listBenchmarks = (projectId: string, agentId: string) => `/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}/benchmarks`, ); +export const listBenchmarkCases = (projectId: string, agentId: string, benchmarkId: string) => + apiFetch( + `/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` + + `/benchmarks/${encodeURIComponent(benchmarkId)}/cases`, + ); + +export const listBenchmarkCaseFiles = ( + projectId: string, + agentId: string, + benchmarkId: string, + caseId: string, + path: string, +) => + apiFetch( + `/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` + + `/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}/files`, + { query: { path } }, + ); + +export const benchmarkCaseFileUrl = ( + projectId: string, + agentId: string, + benchmarkId: string, + caseId: string, + path: string, + options?: { download?: boolean; preview?: boolean }, +): string => { + const base = + `/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` + + `/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}` + + `/files/content?path=${encodeURIComponent(path)}`; + return ( + base + + (options?.download ? "&download=1" : "") + + (options?.preview && !options.download ? "&preview=1" : "") + ); +}; + // Agent State snapshot export / import ------------------------------------------------------ /** Snapshot bundle (tar.gz) download URL: the server sets Content-Disposition attachment, usable directly in . */ diff --git a/packages/web/src/features/benchmark/benchmark-metrics.ts b/packages/web/src/features/benchmark/benchmark-metrics.ts index 8898f69..ef2d42a 100644 --- a/packages/web/src/features/benchmark/benchmark-metrics.ts +++ b/packages/web/src/features/benchmark/benchmark-metrics.ts @@ -1,32 +1,17 @@ /** - * Metric switching and per-model series grouping for the Benchmark center chart (pure functions, - * easy to unit test): the chart can switch between the score / cost / duration metrics (sharing - * the same time axis), and is grouped into series by each evaluation's (provider, modelId) — one - * color per series, legend by model. score is always present; - * cost / durationMs are optional — missing values are **skipped points**: neither drawn nor - * connected, so the line breaks at the gap (lineSegments splits value-bearing indices into - * contiguous segments). + * Score-only data helpers and per-runtime series grouping for the Benchmark center chart. + * Series share one time axis and use each Evaluation's authoritative stored Score. */ -export type BenchmarkMetric = "score" | "cost" | "duration"; - -export const BENCHMARK_METRICS: readonly BenchmarkMetric[] = ["score", "cost", "duration"]; - -/** Minimal evaluation shape needed to read a metric (BenchmarkEvaluation is a superset). */ +/** Minimal Evaluation shape needed to read Score (BenchmarkEvaluation is a superset). */ export interface MetricSourceLike { score: number; - cost?: number; - durationMs?: number; } -/** Each evaluation's value under the selected metric; missing (cost / durationMs not recorded) is null (skipped point). */ -export function metricValues( - evaluations: readonly MetricSourceLike[], - metric: BenchmarkMetric, -): (number | null)[] { +/** Each Evaluation's Score; non-finite malformed values become chart gaps defensively. */ +export function scoreValues(evaluations: readonly MetricSourceLike[]): (number | null)[] { return evaluations.map((e) => { - const v = metric === "score" ? e.score : metric === "cost" ? e.cost : e.durationMs; - return typeof v === "number" && Number.isFinite(v) ? v : null; + return typeof e.score === "number" && Number.isFinite(e.score) ? e.score : null; }); } @@ -63,37 +48,36 @@ export function metricMax(values: readonly (number | null)[]): number { /** Minimal evaluation shape needed for series grouping (BenchmarkEvaluation is a superset). */ export interface ModelRefLike { - provider?: string; modelId?: string; + thinkingLevel?: string; } -/** One chart series: the set of evaluations sharing the same (provider, modelId) (older records with no model tag are grouped into a single series). */ +/** One chart series: the set of evaluations sharing the same (modelId, thinkingLevel). */ export interface EvaluationSeries { /** Grouping key (internal grouping only, not used as an id; empty string for untagged model). */ key: string; - provider?: string; modelId?: string; + thinkingLevel?: string; /** Matching evaluation indices: global time-axis positions, shared across all series on the same x-axis. */ indices: number[]; } /** - * Groups evaluations into series by the model they carry: models are ordered by first - * appearance (color is picked from SERIES_COLORS by series index, so color follows the model and - * doesn't change with filtering); older records with no model tag are grouped into a trailing - * unnamed series (shown in gray, labeled "untagged model"). + * Groups evaluations into series by model ID and thinking level, deliberately ignoring provider. + * Runtime combinations are ordered by first appearance (color is picked from SERIES_COLORS by + * series index); defensive records with no model tag are grouped into a trailing unnamed series. */ export function modelSeries(evaluations: readonly ModelRefLike[]): EvaluationSeries[] { const map = new Map(); evaluations.forEach((e, index) => { const labeled = e.modelId !== undefined && e.modelId !== ""; - const key = labeled ? `${e.provider ?? ""}\u0000${e.modelId}` : ""; + const key = labeled ? `${e.modelId}\u0000${e.thinkingLevel ?? ""}` : ""; let series = map.get(key); if (!series) { series = { key, - ...(labeled && e.provider !== undefined ? { provider: e.provider } : {}), ...(labeled && e.modelId !== undefined ? { modelId: e.modelId } : {}), + ...(labeled && e.thinkingLevel !== undefined ? { thinkingLevel: e.thinkingLevel } : {}), indices: [], }; map.set(key, series); @@ -104,12 +88,11 @@ export function modelSeries(evaluations: readonly ModelRefLike[]): EvaluationSer return [...all.filter((x) => x.key !== ""), ...all.filter((x) => x.key === "")]; } -/** A series' value sequence under the selected metric: indices outside this series are null (skipped point), keeping the global time axis. */ +/** A series' Score sequence; indices outside this series are null, keeping the global time axis. */ export function seriesValues( evaluations: readonly (MetricSourceLike & ModelRefLike)[], series: EvaluationSeries, - metric: BenchmarkMetric, ): (number | null)[] { const own = new Set(series.indices); - return metricValues(evaluations, metric).map((v, i) => (own.has(i) ? v : null)); + return scoreValues(evaluations).map((v, i) => (own.has(i) ? v : null)); } diff --git a/packages/web/src/features/benchmark/benchmark-page.tsx b/packages/web/src/features/benchmark/benchmark-page.tsx index 2f64ff3..4e0094e 100644 --- a/packages/web/src/features/benchmark/benchmark-page.tsx +++ b/packages/web/src/features/benchmark/benchmark-page.tsx @@ -1,21 +1,18 @@ /** * Benchmark page (read-only display): * the left directory lists Benchmarks grouped by Agent (the scoreboard is only fetched once - * expanded); the right side shows the selected Benchmark's title info, a chart (switches between - * score / cost / duration metrics on the same time axis; **grouped into series by the model each - * evaluation carries** — the model isn't part of benchmark_config, each evaluation carries its - * own — one color per series plus a legend, with older untagged records shown as a gray series; - * missing values are skipped points, breaking the line; reuses the usage center's ChartFrame - * coordinate system) and an evaluation detail table (includes a model column; rows expand to - * show the evaluation summary — title and body shown separately — and per-case scores, and case - * rows further expand to show the raw results of each run, with a Session link straight to that - * Session's trace observability). + * expanded); the right side shows the selected Benchmark's title info, a Score-only chart grouped + * into series by each Evaluation's model ID and thinking level, and an evaluation detail table + * with separate model ID and thinking-level columns. Rows expand to show the evaluation summary + * and per-case scores, and Case rows further expand to show the raw results of each Run with a + * Session link. * With a ?agentId= deep link, only the target Agent is expanded by default. */ import { useEffect, useState } from "react"; import { Link, useSearchParams } from "react-router"; import type { BenchmarkCaseScore, + BenchmarkCaseSummary, BenchmarkEvaluation, BenchmarkSummary, } from "@prismshadow/penguin-server/api"; @@ -29,22 +26,22 @@ import { useTheme } from "../../state/theme"; import type { Currency } from "../../state/theme"; import { AgentAvatar } from "../../components/ui/agent-avatar"; import { Chevron } from "../../components/ui/chevron"; -import { Segmented } from "../../components/ui/segmented"; import { Truncated } from "../../components/ui/truncated"; import { EmptyState } from "../../components/ui/empty-state"; +import { Modal } from "../../components/ui/modal"; import { SkeletonList } from "../../components/ui/skeleton"; -import { providerInfo } from "@prismshadow/penguin-core/model-catalog"; import { seriesColor } from "../../lib/category-colors"; import { makeGeom } from "../usage/chart-geom"; import { ChartFrame, useChartWidth } from "../usage/chart-svg"; import { lineSegments, metricMax, - metricValues, modelSeries, + scoreValues, seriesValues, } from "./benchmark-metrics"; -import type { BenchmarkMetric, EvaluationSeries } from "./benchmark-metrics"; +import type { EvaluationSeries } from "./benchmark-metrics"; +import { BenchmarkStatementBrowser } from "./benchmark-statement-browser"; interface Selection { agentId: string; @@ -144,61 +141,31 @@ function AgentNode({ ); } -/** Display label for a metric (shared by the Segmented options and the chart title; S is a live binding, must be read during render). */ -function metricLabel(metric: BenchmarkMetric): string { - return metric === "score" - ? S.benchmark.colScore - : metric === "cost" - ? S.common.cost - : S.benchmark.colDuration; -} - -/** Display format for a metric value (shared by y-axis ticks and the tooltip): score / cost / duration each have their own formatting rule. */ -function formatMetric(metric: BenchmarkMetric, v: number, currency: Currency): string { - return metric === "score" - ? formatScore(v) - : metric === "cost" - ? formatMoney(v, currency) - : humanizeDuration(v); -} - /** - * Metric-over-time line chart (cloned from the usage center's TrendChart: area + line + data - * points, sharing the ChartFrame coordinate system). **Grouped into series** by the model each - * evaluation carries: one color per series (SERIES_COLORS is a fixed color sequence, color - * follows the model; older records with no model tag get a gray series), all series share the - * same time axis and y-axis range. Missing values within a series (indices outside this series, - * or cost / durationMs not recorded) are **skipped points** — lineSegments splits the - * value-bearing indices into segments, drawing area + line + data points within each segment and - * breaking between segments (a single-point segment draws only a point). ChartFrame's x-axis - * labels take slice(5) of dates: passing `yyyy-MM-dd HH:mm` displays as `MM-dd HH:mm`. A single - * evaluation is still drawn; no evaluations falls back to an empty state. + * Score-over-time line chart. Every Run, Case, and Evaluation uses the fixed 0..100 scale. + * Evaluations remain grouped by model ID and thinking level so a runtime change stays visible + * without adding other metric modes. */ -function MetricTrendChart({ +function ScoreTrendChart({ evaluations, series, - metric, - currency, }: { evaluations: BenchmarkEvaluation[]; series: EvaluationSeries[]; - metric: BenchmarkMetric; - currency: Currency; }) { const [hover, setHover] = useState(null); const [ref, width] = useChartWidth(); - const values = metricValues(evaluations, metric); - const geom = makeGeom(evaluations.length, metricMax(values), width); + const values = scoreValues(evaluations); + const geom = makeGeom(evaluations.length, Math.max(metricMax(values), 100), width); const dates = evaluations.map((e) => formatDateTime(e.time)); - const baseY = geom.y(0); return (
{width > 0 && ( formatMetric(metric, v, currency)} + fmtY={formatScore} dates={dates} hover={hover} onHover={setHover} @@ -209,18 +176,20 @@ function MetricTrendChart({ <>

{formatDateTime(e.time)}

- {v === null ? "—" : formatMetric(metric, v, currency)} + {v === null ? "—" : formatScore(v)} {e.version !== undefined && ( v{e.version} )}

- {e.modelId &&

{e.modelId}

} +

+ {e.modelId} · {e.thinkingLevel} +

); }} > {series.map((s, si) => { - const segments = lineSegments(seriesValues(evaluations, s, metric)); + const segments = lineSegments(seriesValues(evaluations, s)); return ( `${j === 0 ? "M" : "L"}${geom.x(p.index)},${geom.y(p.value)}`) .join(" "); - const area = `${line} L${geom.x(seg[seg.length - 1]!.index)},${baseY} L${geom.x(seg[0]!.index)},${baseY} Z`; return ( - {/* Area fill: line closed to the baseline, low opacity reinforces the trend's sense of "volume" (no area for single-point segments) */} - {seg.length > 1 && ( - - )} {seg.length > 1 && ( =2 series; a single series' identity is carried by - * the detail table's model column and the hover tooltip instead) + the line chart. When the same - * modelId coexists across providers, the legend appends the provider's display name to - * disambiguate. Mounted under a keyed container per Benchmark: switching Benchmarks resets back - * to "score". + * Score chart + runtime legend. Provider is deliberately not part of chart identity. */ -function TrendSection({ - evaluations, - currency, -}: { - evaluations: BenchmarkEvaluation[]; - currency: Currency; -}) { - const [metric, setMetric] = useState("score"); +function TrendSection({ evaluations }: { evaluations: BenchmarkEvaluation[] }) { const series = modelSeries(evaluations); - const ids = series.map((s) => s.modelId).filter((v): v is string => v !== undefined); - const dupIds = new Set(ids.filter((id, i) => ids.indexOf(id) !== i)); const labelOf = (s: EvaluationSeries): string => { if (!s.modelId) return S.benchmark.legendUnlabeled; - if (!dupIds.has(s.modelId)) return s.modelId; - const provider = s.provider ? (providerInfo(s.provider)?.label ?? s.provider) : ""; - return provider ? `${s.modelId} · ${provider}` : s.modelId; + return s.thinkingLevel ? `${s.modelId} · ${s.thinkingLevel}` : s.modelId; }; return (
-
-

- {S.benchmark.trendTitle(metricLabel(metric))} -

-
- -
-
+

+ {S.benchmark.trendTitle(S.benchmark.colScore)} +

{series.length >= 2 && (
{series.map((s, i) => ( )} - +
); } const CELL = "px-3 py-2"; -/** One evaluation record: main row (time/version/total score/cost/duration) + a sub-table of per-case scores that expands on click. */ +/** One evaluation record: main row + a sub-table of per-Case scores that expands on click. */ function EvaluationRow({ agentId, evaluation, + caseTitles, + onOpenCase, currency, }: { agentId: string; evaluation: BenchmarkEvaluation; + caseTitles: ReadonlyMap; + onOpenCase: (caseId: string) => void; currency: Currency; }) { const [open, setOpen] = useState(false); @@ -376,7 +304,10 @@ function EvaluationRow({ className={`${CELL} max-w-40 truncate font-mono text-xs text-gray-500 dark:text-gray-400`} title={evaluation.provider} > - {evaluation.modelId ?? "—"} + {evaluation.modelId} + + + {evaluation.thinkingLevel} {formatScore(evaluation.score)} @@ -390,11 +321,8 @@ function EvaluationRow({ {open && ( - - {/* Evaluation summary (title + body shown separately; the generating side always - writes both, but the display side tolerates missing values — with an old-style - single-paragraph summary only, it's still shown as usual, prefixed with the - "Evaluation Summary" label). */} + + {/* Evaluation summary title and body are displayed separately when present. */} {(evaluation.summaryTitle || evaluation.summary) && (
{evaluation.summaryTitle ? ( @@ -423,7 +351,14 @@ function EvaluationRow({ {evaluation.cases.map((c) => ( - + ))} @@ -449,33 +384,54 @@ function SessionLink({ agentId, sessionId }: { agentId: string; sessionId?: stri } /** - * Score row for one case: the case-level metrics = the average of its runs (already computed by - * the server, trust its values). With runs[] present, the row can expand to show the raw results - * of each run (#index + score / cost / duration / Session link); with the old format lacking - * runs, it's not expandable and the case-level single Session link is used as before. + * Score row for one Case: stored Case averages are authoritative. Expanding shows raw Run + * results; the UI never recomputes averages. */ function CaseRow({ agentId, caseScore: c, + title, + onOpenCase, currency, }: { agentId: string; caseScore: BenchmarkCaseScore; + title?: string; + onOpenCase?: (caseId: string) => void; currency: Currency; }) { const [open, setOpen] = useState(false); - const runs = c.runs ?? []; - const expandable = runs.length > 0; + const runs = c.runs; return ( <> setOpen((v) => !v) : undefined} - className={`text-xs ${expandable ? "cursor-pointer transition-colors duration-150 hover:bg-gray-100/70 dark:hover:bg-gray-800/40" : ""}`} + onClick={() => setOpen((v) => !v)} + className="cursor-pointer text-xs transition-colors duration-150 hover:bg-gray-100/70 dark:hover:bg-gray-800/40" > - - - {expandable && } - {c.case} + + + + + {onOpenCase ? ( + + ) : ( + + {title ?? c.case} + + )} + {title && title !== c.case && ( + {c.case} + )} + {formatScore(c.score)} @@ -486,7 +442,7 @@ function CaseRow({ {c.durationMs !== undefined ? humanizeDuration(c.durationMs) : "—"} - + — {open && @@ -513,6 +469,51 @@ function CaseRow({ ); } +function CasesSection({ + cases, + error, + onOpenCase, +}: { + cases: BenchmarkCaseSummary[] | null; + error: string | null; + onOpenCase: (caseId: string) => void; +}) { + return ( +
+

{S.benchmark.cases}

+
+ {error &&

{error}

} + {!cases && !error &&

{S.common.loading}

} + {cases?.map((item) => { + return ( + + ); + })} +
+
+ ); +} + export function BenchmarkPage() { useDocumentTitle(S.benchmark.title); const { currentProject, agents, agentsLoading } = useProject(); @@ -522,17 +523,42 @@ export function BenchmarkPage() { const [searchParams] = useSearchParams(); const focusAgentId = searchParams.get("agentId"); const [selection, setSelection] = useState(null); + const [caseStatements, setCaseStatements] = useState(null); + const [caseError, setCaseError] = useState(null); + const [openCaseId, setOpenCaseId] = useState(null); // Clear the selection when the Project changes. useEffect(() => { setSelection(null); }, [projectId]); + useEffect(() => { + setCaseStatements(null); + setCaseError(null); + setOpenCaseId(null); + if (!projectId || !selection) return; + let cancelled = false; + api + .listBenchmarkCases(projectId, selection.agentId, selection.benchmark.id) + .then((data) => { + if (!cancelled) setCaseStatements(data.cases); + }) + .catch((error: unknown) => { + if (!cancelled) setCaseError(apiErrorText(error)); + }); + return () => { + cancelled = true; + }; + }, [projectId, selection]); + if (!projectId) return null; const bm = selection?.benchmark ?? null; - // Chart uses ascending time order (the scoreboard is already ordered, this sort is defensive); the detail table shows newest first. - const evaluations = bm ? [...bm.evaluations].sort((a, b) => a.time.localeCompare(b.time)) : []; + // The Scoreboard append order is the evaluation sequence. Preserve it even when a malformed + // timestamp would otherwise reorder Agent versions; the detail table shows that sequence newest first. + const evaluations = bm ? [...bm.evaluations] : []; + const caseTitles = new Map(caseStatements?.map((item) => [item.id, item.title]) ?? []); + const openCase = caseStatements?.find((item) => item.id === openCaseId) ?? null; return (
@@ -564,9 +590,7 @@ export function BenchmarkPage() { {selection && bm ? ( // Changing the key on Benchmark switch resets expand state (a detail row's open doesn't linger across Benchmarks).
- {/* Title row: title + case count (the model isn't part of config — each evaluation - carries its own, see the chart legend and the detail table's model column) + - description */} + {/* Runtime belongs to each Evaluation and is shown in the detail table. */}

{bm.title}

@@ -577,23 +601,26 @@ export function BenchmarkPage() { )}
+ + {evaluations.length === 0 ? ( ) : ( <> - +

{S.benchmark.evaluations}

- +
+ @@ -605,6 +632,8 @@ export function BenchmarkPage() { key={i} agentId={selection.agentId} evaluation={ev} + caseTitles={caseTitles} + onOpenCase={setOpenCaseId} currency={currency} /> ))} @@ -614,6 +643,21 @@ export function BenchmarkPage() { )} + {openCase && ( + setOpenCaseId(null)} + > + + + )} ) : ( diff --git a/packages/web/src/features/benchmark/benchmark-statement-browser.tsx b/packages/web/src/features/benchmark/benchmark-statement-browser.tsx new file mode 100644 index 0000000..86f4077 --- /dev/null +++ b/packages/web/src/features/benchmark/benchmark-statement-browser.tsx @@ -0,0 +1,367 @@ +import { useCallback, useEffect, useRef, useState } from "react"; +import type { + BenchmarkCaseSummary, + WorkspaceFileEntry, + WorkspaceFilesResponse, +} from "@prismshadow/penguin-server/api"; +import ReactMarkdown from "react-markdown"; +import type { Components } from "react-markdown"; +import remarkGfm from "remark-gfm"; +import * as api from "../../api/endpoints"; +import { apiErrorText } from "../../lib/api-error"; +import { formatBytes } from "../../lib/format"; +import { S } from "../../lib/strings"; +import { SkeletonList } from "../../components/ui/skeleton"; +import { CodeBlock } from "../chat/code-block"; + +const TEXT_EXTS = new Set([ + "txt", + "md", + "json", + "js", + "mjs", + "cjs", + "ts", + "tsx", + "jsx", + "py", + "sh", + "bash", + "yaml", + "yml", + "toml", + "css", + "html", + "htm", + "csv", + "log", + "xml", + "ini", + "conf", + "sql", + "svg", +]); +const IMAGE_EXTS = new Set(["png", "jpg", "jpeg", "gif", "webp"]); +const EXTERNAL_REF_RE = /^[a-z][a-z0-9+.-]*:/i; +const HIGHLIGHT_LIMIT = 64 * 1024; + +interface Preview { + path: string; + name: string; + kind: "text" | "md" | "image" | "pdf" | "unsupported"; + content?: string; + truncated?: boolean; + loading?: boolean; + error?: string; +} + +interface Props { + projectId: string; + agentId: string; + benchmarkId: string; + caseSummary: BenchmarkCaseSummary; +} + +function extOf(name: string): string { + const index = name.lastIndexOf("."); + return index >= 0 ? name.slice(index + 1).toLowerCase() : name.toLowerCase(); +} + +function joinPath(dir: string, name: string): string { + return dir === "" ? name : `${dir}/${name}`; +} + +function dirOf(filePath: string): string { + return filePath.includes("/") ? filePath.slice(0, filePath.lastIndexOf("/")) : ""; +} + +function resolveRelative(baseDir: string, ref: string): string { + const out = ref.startsWith("/") || baseDir === "" ? [] : baseDir.split("/"); + for (const segment of ref.split("/")) { + if (segment === "" || segment === ".") continue; + if (segment === "..") out.pop(); + else out.push(segment); + } + return out.join("/"); +} + +function languageFor(name: string): string { + const ext = extOf(name); + return ( + { + json: "json", + md: "markdown", + js: "javascript", + mjs: "javascript", + cjs: "javascript", + jsx: "jsx", + ts: "typescript", + tsx: "tsx", + py: "python", + sh: "shellscript", + bash: "shellscript", + yaml: "yaml", + yml: "yaml", + toml: "toml", + css: "css", + html: "html", + htm: "html", + xml: "xml", + sql: "sql", + svg: "xml", + }[ext] ?? "text" + ); +} + +export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, caseSummary }: Props) { + const [path, setPath] = useState(""); + const [listing, setListing] = useState(null); + const [listError, setListError] = useState(null); + const [preview, setPreview] = useState(null); + const initialReadmeOpened = useRef(false); + const previewRequest = useRef(0); + + const fileUrl = useCallback( + (filePath: string, options?: { download?: boolean; preview?: boolean }) => + api.benchmarkCaseFileUrl(projectId, agentId, benchmarkId, caseSummary.id, filePath, options), + [projectId, agentId, benchmarkId, caseSummary.id], + ); + + const previewPath = useCallback( + async (filePath: string) => { + const request = ++previewRequest.current; + const name = filePath.includes("/") + ? filePath.slice(filePath.lastIndexOf("/") + 1) + : filePath; + const ext = extOf(name); + if (IMAGE_EXTS.has(ext)) { + setPreview({ path: filePath, name, kind: "image" }); + return; + } + if (ext === "pdf") { + setPreview({ path: filePath, name, kind: "pdf" }); + return; + } + const isMarkdown = ext === "md"; + if (!TEXT_EXTS.has(ext)) { + setPreview({ path: filePath, name, kind: "unsupported" }); + return; + } + setPreview({ path: filePath, name, kind: isMarkdown ? "md" : "text", loading: true }); + try { + const response = await fetch(fileUrl(filePath, { preview: true }), { + credentials: "same-origin", + }); + if (!response.ok) throw new Error(String(response.status)); + const content = await response.text(); + if (request !== previewRequest.current) return; + setPreview({ + path: filePath, + name, + kind: isMarkdown ? "md" : "text", + content, + truncated: response.headers.get("x-content-truncated") === "1", + }); + } catch (error) { + if (request !== previewRequest.current) return; + setPreview({ + path: filePath, + name, + kind: isMarkdown ? "md" : "text", + error: apiErrorText(error), + }); + } + }, + [fileUrl], + ); + + useEffect(() => { + setListing(null); + setListError(null); + let cancelled = false; + api + .listBenchmarkCaseFiles(projectId, agentId, benchmarkId, caseSummary.id, path) + .then((data) => { + if (cancelled) return; + setListing(data); + if (path === "" && !initialReadmeOpened.current) { + initialReadmeOpened.current = true; + const readme = data.entries.find( + (entry) => entry.kind === "file" && entry.name.toLowerCase() === "readme.md", + ); + if (readme) void previewPath(readme.name); + } + }) + .catch((error: unknown) => { + if (!cancelled) setListError(apiErrorText(error)); + }); + return () => { + cancelled = true; + }; + }, [projectId, agentId, benchmarkId, caseSummary.id, path, previewPath]); + + const crumbs = path === "" ? [] : path.split("/"); + const downloadUrl = preview ? fileUrl(preview.path, { download: true }) : null; + + const openEntry = (entry: WorkspaceFileEntry) => { + if (entry.kind === "dir") { + setPath(joinPath(path, entry.name)); + return; + } + void previewPath(joinPath(path, entry.name)); + }; + + const markdownComponents: Components = { + img: ({ src, alt }) => ( + {alt + ), + a: ({ href, children }) => { + if (typeof href !== "string" || href.startsWith("#")) return {children}; + if (EXTERNAL_REF_RE.test(href)) { + return ( + + {children} + + ); + } + const target = resolveRelative(dirOf(preview?.path ?? ""), href); + return ( + { + event.preventDefault(); + void previewPath(target); + }} + > + {children} + + ); + }, + }; + + return ( +
+ + +
+
+
+

+ {preview?.path ?? caseSummary.id} +

+
+ {S.benchmark.maxScore("100")} + {downloadUrl && preview && ( + + {S.files.download} + + )} +
+
+ {!preview ? ( +

{S.benchmark.statementUnavailable}

+ ) : preview.loading ? ( + + ) : preview.error ? ( +

{preview.error}

+ ) : preview.kind === "image" ? ( + {preview.name} + ) : preview.kind === "pdf" ? ( +
{S.common.time} {S.benchmark.colVersion} {S.benchmark.colModel}{S.benchmark.colThinkingLevel} {S.benchmark.colScore} {S.common.cost} {S.benchmark.colDuration}