Improve benchmark case details and score chart scaling (#146)

Co-authored-by: Yaowei Zheng <hiyouga@buaa.edu.cn>
This commit is contained in:
Jingzhe Xu
2026-07-31 20:42:32 +08:00
committed by GitHub
parent 5308002d72
commit 740f101242
19 changed files with 519 additions and 212 deletions
@@ -0,0 +1,9 @@
# Evaluation Center Case details and Score charts
Case details now separate the task materials visible to the Target Agent from the scoring rubric available to project members, while keeping both file roots path-confined. Score charts use a padded dynamic axis without discarding authoritative stored values, and the benchmark Skills use YAML folded scalars for Scoreboard summaries.
## Details
- Case details show task materials and scoring rubrics as separate file groups. Rubrics remain hidden from the Target Agent.
- Score charts pad the observed in-range values, clamp the displayed axis to 0–100, and preserve stored finite values for plotting and tooltips.
- Benchmark design and optimization Skills write summary fields as YAML folded scalars and parse the complete Scoreboard after appending an Evaluation.
+2
View File
@@ -2,4 +2,6 @@
Changes since v0.1.5. The version number is assigned at release, when this folder is renamed.
- [2026-07-31] Evaluation Center: Case details separate Target Agent task materials from project-member-visible scoring rubrics, Score charts use a padded dynamic axis without discarding stored values, and the benchmark Skills write YAML-safe Scoreboard summaries. ([details](2026-07-31-evaluation-center-case-details.md))
- [2026-07-30] Web App: the main conversation's file summary moves to the completed Task boundary — one card per Task scanning all of its assistant text, nested agent conversations keep their per-message summaries, and file-existence caching stops retaining negative results so a later Task can surface a newly created path. ([details](2026-07-30-file-summary-task-boundary.md))
+2
View File
@@ -1292,6 +1292,8 @@ export interface BenchmarksResponse {
benchmarks: BenchmarkSummary[];
}
export type CaseMaterial = "statement" | "rubric";
/** Public Benchmark Case metadata. Rubric and Gold content are never included. */
export interface BenchmarkCaseSummary {
id: string;
+61 -47
View File
@@ -4,16 +4,72 @@
* GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases
* GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/files
* GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/files/content
* GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/rubric/files
* GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/rubric/files/content
* Returns the Agent's Benchmark list (title/description from benchmark_config.toml)
* along with the evaluations[] from scoreboard.yaml.
*/
import { Hono } from "hono";
import { Hono, type Context } from "hono";
import type { AppEnv } from "../../auth/middleware.js";
import type { AppDeps } from "../../app.js";
import type { CaseMaterial } from "../../api/types.js";
import { requireValidId } from "../validate.js";
const TEXT_PREVIEW_BYTES = 256 * 1024;
function listCaseFiles(deps: AppDeps, material: CaseMaterial) {
return async (c: Context<AppEnv>) => {
const projectId = requireValidId(c, "projectId");
const agentId = requireValidId(c, "agentId");
const benchmarkId = requireValidId(c, "benchmarkId");
const caseId = requireValidId(c, "caseId");
deps.projectService.requireProjectAccess(c.var.user.userId, projectId);
await deps.agentConfigService.requireExists(projectId, agentId);
return c.json(
await deps.benchmarks.listCaseFiles(
projectId,
agentId,
benchmarkId,
caseId,
c.req.query("path") ?? "",
material,
),
);
};
}
function readCaseFile(deps: AppDeps, material: CaseMaterial) {
return async (c: Context<AppEnv>) => {
const projectId = requireValidId(c, "projectId");
const agentId = requireValidId(c, "agentId");
const benchmarkId = requireValidId(c, "benchmarkId");
const caseId = requireValidId(c, "caseId");
deps.projectService.requireProjectAccess(c.var.user.userId, projectId);
await deps.agentConfigService.requireExists(projectId, agentId);
const download = c.req.query("download") === "1";
const boundedPreview = !download && c.req.query("preview") === "1";
const { data, fileName, contentType, scriptable, truncated } =
await deps.benchmarks.readCaseFile(
projectId,
agentId,
benchmarkId,
caseId,
c.req.query("path") ?? "",
material,
boundedPreview ? { maxBytes: TEXT_PREVIEW_BYTES } : undefined,
);
return new Response(new Uint8Array(data), {
status: 200,
headers: {
"Content-Type": !download && scriptable ? "text/plain; charset=utf-8" : contentType,
"Content-Disposition": `${download ? "attachment" : "inline"}; filename*=UTF-8''${encodeURIComponent(fileName)}`,
"X-Content-Type-Options": "nosniff",
...(truncated ? { "X-Content-Truncated": "1" } : {}),
},
});
};
}
export function benchmarksRoutes(deps: AppDeps): Hono<AppEnv> {
const app = new Hono<AppEnv>();
@@ -34,52 +90,10 @@ export function benchmarksRoutes(deps: AppDeps): Hono<AppEnv> {
return c.json(await deps.benchmarks.listCases(projectId, agentId, benchmarkId));
});
app.get("/:benchmarkId/cases/:caseId/files", async (c) => {
const projectId = requireValidId(c, "projectId");
const agentId = requireValidId(c, "agentId");
const benchmarkId = requireValidId(c, "benchmarkId");
const caseId = requireValidId(c, "caseId");
deps.projectService.requireProjectAccess(c.var.user.userId, projectId);
await deps.agentConfigService.requireExists(projectId, agentId);
return c.json(
await deps.benchmarks.listCaseFiles(
projectId,
agentId,
benchmarkId,
caseId,
c.req.query("path") ?? "",
),
);
});
app.get("/:benchmarkId/cases/:caseId/files/content", async (c) => {
const projectId = requireValidId(c, "projectId");
const agentId = requireValidId(c, "agentId");
const benchmarkId = requireValidId(c, "benchmarkId");
const caseId = requireValidId(c, "caseId");
deps.projectService.requireProjectAccess(c.var.user.userId, projectId);
await deps.agentConfigService.requireExists(projectId, agentId);
const download = c.req.query("download") === "1";
const boundedPreview = !download && c.req.query("preview") === "1";
const { data, fileName, contentType, scriptable, truncated } =
await deps.benchmarks.readCaseFile(
projectId,
agentId,
benchmarkId,
caseId,
c.req.query("path") ?? "",
boundedPreview ? { maxBytes: TEXT_PREVIEW_BYTES } : undefined,
);
return new Response(new Uint8Array(data), {
status: 200,
headers: {
"Content-Type": !download && scriptable ? "text/plain; charset=utf-8" : contentType,
"Content-Disposition": `${download ? "attachment" : "inline"}; filename*=UTF-8''${encodeURIComponent(fileName)}`,
"X-Content-Type-Options": "nosniff",
...(truncated ? { "X-Content-Truncated": "1" } : {}),
},
});
});
app.get("/:benchmarkId/cases/:caseId/files", listCaseFiles(deps, "statement"));
app.get("/:benchmarkId/cases/:caseId/files/content", readCaseFile(deps, "statement"));
app.get("/:benchmarkId/cases/:caseId/rubric/files", listCaseFiles(deps, "rubric"));
app.get("/:benchmarkId/cases/:caseId/rubric/files/content", readCaseFile(deps, "rubric"));
return app;
}
@@ -25,6 +25,7 @@ import type {
BenchmarkRunScore,
BenchmarkSummary,
BenchmarksResponse,
CaseMaterial,
WorkspaceFilesResponse,
} from "../api/types.js";
import type {
@@ -235,7 +236,13 @@ export class BenchmarkService {
.sort((a, b) => a.name.localeCompare(b.name))) {
const fallback: BenchmarkCaseSummary = { id: entry.name, title: entry.name };
try {
const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, entry.name);
const statementDir = await this.caseMaterialRoot(
projectId,
agentId,
benchmarkId,
entry.name,
"statement",
);
const realReadme = await fs.realpath(path.join(statementDir, "README.md"));
if (!isWithin(statementDir, realReadme)) throw new Error("README escapes Statement");
cases.push({
@@ -255,9 +262,16 @@ export class BenchmarkService {
benchmarkId: string,
caseId: string,
rel: string,
material: CaseMaterial,
): Promise<WorkspaceFilesResponse> {
const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, caseId);
return this.workspaceFiles.list(statementDir, rel);
const materialRoot = await this.caseMaterialRoot(
projectId,
agentId,
benchmarkId,
caseId,
material,
);
return this.workspaceFiles.list(materialRoot, rel);
}
async readCaseFile(
@@ -266,37 +280,45 @@ export class BenchmarkService {
benchmarkId: string,
caseId: string,
rel: string,
material: CaseMaterial,
options?: WorkspaceFileReadOptions,
): Promise<WorkspaceFileContent> {
const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, caseId);
return this.workspaceFiles.read(statementDir, rel, options);
const materialRoot = await this.caseMaterialRoot(
projectId,
agentId,
benchmarkId,
caseId,
material,
);
return this.workspaceFiles.read(materialRoot, rel, options);
}
private async statementRoot(
private async caseMaterialRoot(
projectId: string,
agentId: string,
benchmarkId: string,
caseId: string,
material: CaseMaterial,
): Promise<string> {
const benchDir = path.join(benchmarksDir(this.root, projectId, agentId), benchmarkId);
const caseDir = path.join(benchDir, caseId);
const statementDir = path.join(caseDir, "statement");
const materialRoot = path.join(caseDir, material);
try {
const [realBenchDir, realCaseDir, realStatementDir] = await Promise.all([
const [realBenchDir, realCaseDir, realMaterialRoot] = await Promise.all([
fs.realpath(benchDir),
fs.realpath(caseDir),
fs.realpath(statementDir),
fs.realpath(materialRoot),
]);
if (
!isWithin(realBenchDir, realCaseDir) ||
path.dirname(realStatementDir) !== realCaseDir ||
path.basename(realStatementDir) !== "statement"
path.dirname(realMaterialRoot) !== realCaseDir ||
path.basename(realMaterialRoot) !== material
) {
throw new Error("Statement path is not canonical");
throw new Error("Case material is not canonical");
}
return realStatementDir;
return realMaterialRoot;
} catch {
throw new HttpError(404, "not_found", "Public Case Statement does not exist.");
throw new HttpError(404, "not_found", `Case ${material} does not exist.`);
}
}
+34
View File
@@ -69,6 +69,7 @@ describe("benchmarks api", () => {
await fs.mkdir(path.join(dir, "CASE-001-excel-task", "statement", "assets"), {
recursive: true,
});
await fs.mkdir(path.join(dir, "CASE-001-excel-task", "rubric"), { recursive: true });
await fs.mkdir(path.join(dir, "CASE-002-web-task", "statement"), { recursive: true });
await fs.mkdir(path.join(dir, "CASE-002-web-task", "rubric"), { recursive: true });
await fs.writeFile(
@@ -91,6 +92,16 @@ describe("benchmarks api", () => {
"x".repeat(300 * 1024),
"utf8",
);
await fs.writeFile(
path.join(dir, "CASE-001-excel-task", "rubric", "README.md"),
"# Scoring rubric\n\n- Correct workbook: 100 points",
"utf8",
);
await fs.writeFile(
path.join(dir, "CASE-001-excel-task", "rubric", "expected.json"),
'{"rows": 1}\n',
"utf8",
);
await fs.writeFile(
path.join(dir, "CASE-002-web-task", "statement", "README.md"),
"# Web task\n\nBuild the page.",
@@ -245,6 +256,22 @@ describe("benchmarks api", () => {
expect(statement.headers.get("content-type")).toContain("markdown");
expect(await statement.text()).toBe("# Case 001: Excel cleanup\n\nClean the workbook.");
const rubricFilesBase = `${base}/swe-bench-v2/cases/CASE-001-excel-task/rubric/files`;
const rubricFiles = (await (
await member.get(rubricFilesBase)
).json()) as WorkspaceFilesResponse;
expect(rubricFiles.entries.map((entry) => `${entry.kind}:${entry.name}`)).toEqual([
"file:expected.json",
"file:README.md",
]);
const rubric = await member.get(
`${rubricFilesBase}/content?path=${encodeURIComponent("README.md")}&preview=1`,
);
expect(rubric.status).toBe(200);
expect(rubric.headers.get("content-type")).toContain("markdown");
expect(await rubric.text()).toContain("Correct workbook: 100 points");
const large = await member.get(
`${filesBase}/content?path=${encodeURIComponent("large.txt")}&preview=1`,
);
@@ -262,6 +289,13 @@ describe("benchmarks api", () => {
(await member.get(`${filesBase}/content?path=${encodeURIComponent("../rubric/README.md")}`))
.status,
).toBe(400);
expect(
(
await member.get(
`${rubricFilesBase}/content?path=${encodeURIComponent("../statement/README.md")}`,
)
).status,
).toBe(400);
expect((await outsider.get(`${base}/swe-bench-v2/cases`)).status).toBe(404);
expect((await outsider.get(filesBase)).status).toBe(404);
});
@@ -3,8 +3,8 @@ name: agent-optimization
description: Improve an Agent State through versioned scores and score-linked Traces from a frozen Benchmark.
short_description: Improve an Agent from measured Benchmark results.
short_description_zh: 根据 Benchmark 结果改进 Agent。
version: 7
updated: 2026-07-30T02:51:10Z
version: 8
updated: 2026-07-31T04:10:53Z
---
# Agent Optimization
@@ -106,8 +106,10 @@ Append each complete accepted Candidate Evaluation to `scoreboard.yaml` immediat
provider: <provider>
model_id: <model_id>
thinking_level: <thinking_level>
summary_title: <public title>
summary: <public summary>
summary_title: >-
<public title>
summary: >-
<public summary>
score: <average of the Case scores>
cost: <average of known Case costs, or null when every Case cost is null>
duration_ms: <average of the Case durations>
@@ -123,6 +125,8 @@ Append each complete accepted Candidate Evaluation to `scoreboard.yaml` immediat
session_id: <Test Session id>
```
After writing, parse the complete `scoreboard.yaml` and verify the appended Evaluation before reporting success or continuing.
Every Run and Case score is on the fixed `0..100` scale. Do not write `max_score`. Calculate and write every Case and Evaluation average directly in the Scoreboard: ignore `null` values when averaging cost and write `null` only when all contributing costs are unknown; round `score` averages to two decimal places, `cost` averages to six decimal places, and `duration_ms` averages to the nearest integer. These stored values are authoritative—do not add a server, frontend, script, or consistency check that recomputes or validates them. Do not add an `aggregate` object or use `case_id`, `mean_score`, `mean_cost`, or `mean_duration_ms`. Do not record rejected Candidates in the Scoreboard.
Report the Baseline and every fully evaluated Candidate with its score, version, change, decision, and Test Session ids. For each Candidate, distinguish the acceptance decision from whether its stated hypothesis was supported by the predicted Case behavior. Include the final retained version, stop reason, and known limitations. Never report a score for an Agent State that was not evaluated.
@@ -3,8 +3,8 @@ name: benchmark-design
description: Design and calibrate a multi-Case capability Benchmark and establish a traceable Formal Baseline.
short_description: Design and calibrate an Agent capability Benchmark.
short_description_zh: 设计并校准 Agent 能力评测 Benchmark。
version: 5
updated: 2026-07-29T17:20:58Z
version: 6
updated: 2026-07-31T04:10:53Z
---
# Benchmark Design
@@ -175,8 +175,10 @@ evaluations:
provider: <provider>
model_id: <model_id>
thinking_level: <thinking_level>
summary_title: <public title>
summary: <public summary>
summary_title: >-
<public title>
summary: >-
<public summary>
score: <average of the Case scores>
cost: <average of known Case costs, or null when every Case cost is null>
duration_ms: <average of the Case durations>
@@ -192,6 +194,8 @@ evaluations:
session_id: <Test Session id>
```
After writing, parse the complete `scoreboard.yaml` and verify the appended Evaluation before reporting success or continuing.
Every Run and Case score is on the fixed `0..100` scale. Do not write `max_score`. Calculate and write every Case and Evaluation average directly in the Scoreboard: ignore `null` values when averaging cost and write `null` only when all contributing costs are unknown; round `score` averages to two decimal places, `cost` averages to six decimal places, and `duration_ms` averages to the nearest integer. These stored values are authoritative—do not add a server, frontend, script, or consistency check that recomputes or validates them. Do not add an `aggregate` object or use `case_id`, `mean_score`, `mean_cost`, or `mean_duration_ms`.
Report the Benchmark path, configuration, Agent State version, Evaluation average and Case Run scores, Test Session ids, and known limitations. Include one compact row per Pilot iteration with its score, diagnosed capability gap, difficulty adjustment, and freeze or stop decision.
+10
View File
@@ -300,6 +300,11 @@ describe("librarySkill", () => {
expect(benchmarkDesign).toContain("ignore `null` values when averaging cost");
expect(benchmarkDesign).toContain("stored values are authoritative");
expect(benchmarkDesign).toContain("obtain the current UTC timestamp from the environment");
expect(benchmarkDesignRaw).toMatch(/summary_title:\s*>-\n\s+<public title>/);
expect(benchmarkDesignRaw).toMatch(/summary:\s*>-\n\s+<public summary>/);
expect(benchmarkDesign).toContain(
"parse the complete `scoreboard.yaml` and verify the appended Evaluation",
);
expect(benchmarkDesignRaw).not.toMatch(/\n\s+max_score:/);
expect(benchmarkDesign).toContain(
"Before reading `status`, `score`, or any other protocol field",
@@ -308,6 +313,11 @@ describe("librarySkill", () => {
expect(optimization).toContain("Delegate every evaluation to an `agent-evaluation` subagent");
expect(optimization).toContain("Before reading `status`, `score`, or any other protocol field");
expect(optimization).toContain("Ask the same Evaluator to resend only the clean YAML");
expect(optimizationRaw).toMatch(/summary_title:\s*>-\n\s+<public title>/);
expect(optimizationRaw).toMatch(/summary:\s*>-\n\s+<public summary>/);
expect(optimization).toContain(
"parse the complete `scoreboard.yaml` and verify the appended Evaluation",
);
expect(optimization).toContain(
"A round counts only after one Candidate has a complete valid Evaluation",
);
+22 -6
View File
@@ -23,6 +23,7 @@ import type {
AuthResponse,
BenchmarkCasesResponse,
BenchmarksResponse,
CaseMaterial,
DirListResponse,
FilesStatRequest,
FilesStatResponse,
@@ -538,16 +539,27 @@ export const listBenchmarkCases = (projectId: string, agentId: string, benchmark
`/benchmarks/${encodeURIComponent(benchmarkId)}/cases`,
);
const benchmarkCaseFilesPath = (
projectId: string,
agentId: string,
benchmarkId: string,
caseId: string,
material: CaseMaterial,
) =>
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` +
`/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}` +
`${material === "rubric" ? "/rubric" : ""}/files`;
export const listBenchmarkCaseFiles = (
projectId: string,
agentId: string,
benchmarkId: string,
caseId: string,
path: string,
material: CaseMaterial,
) =>
apiFetch<WorkspaceFilesResponse>(
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` +
`/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}/files`,
benchmarkCaseFilesPath(projectId, agentId, benchmarkId, caseId, material),
{ query: { path } },
);
@@ -557,12 +569,16 @@ export const benchmarkCaseFileUrl = (
benchmarkId: string,
caseId: string,
path: string,
material: CaseMaterial,
options?: { download?: boolean; preview?: boolean },
): string => {
const base =
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` +
`/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}` +
`/files/content?path=${encodeURIComponent(path)}`;
const base = `${benchmarkCaseFilesPath(
projectId,
agentId,
benchmarkId,
caseId,
material,
)}/content?path=${encodeURIComponent(path)}`;
return (
base +
(options?.download ? "&download=1" : "") +
@@ -1,6 +1,7 @@
import { useCallback, useEffect, useRef, useState } from "react";
import type {
BenchmarkCaseSummary,
CaseMaterial,
WorkspaceFileEntry,
WorkspaceFilesResponse,
} from "@prismshadow/penguin-server/api";
@@ -46,6 +47,7 @@ const EXTERNAL_REF_RE = /^[a-z][a-z0-9+.-]*:/i;
const HIGHLIGHT_LIMIT = 64 * 1024;
interface Preview {
material: CaseMaterial;
path: string;
name: string;
kind: "text" | "md" | "image" | "pdf" | "unsupported";
@@ -62,6 +64,14 @@ interface Props {
caseSummary: BenchmarkCaseSummary;
}
interface MaterialGroupProps extends Props {
material: CaseMaterial;
label: string;
hiddenLabel?: string;
defaultOpen?: boolean;
onPreview: (material: CaseMaterial, path: string) => void;
}
function extOf(name: string): string {
const index = name.lastIndexOf(".");
return index >= 0 ? name.slice(index + 1).toLowerCase() : name.toLowerCase();
@@ -113,49 +123,185 @@ function languageFor(name: string): string {
);
}
export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, caseSummary }: Props) {
function MaterialGroup({
projectId,
agentId,
benchmarkId,
caseSummary,
material,
label,
hiddenLabel,
defaultOpen = false,
onPreview,
}: MaterialGroupProps) {
const [open, setOpen] = useState(defaultOpen);
const [path, setPath] = useState("");
const [listing, setListing] = useState<WorkspaceFilesResponse | null>(null);
const [listError, setListError] = useState<string | null>(null);
const [preview, setPreview] = useState<Preview | null>(null);
const initialReadmeOpened = useRef(false);
useEffect(() => {
if (!open) return;
setListing(null);
setListError(null);
let cancelled = false;
api
.listBenchmarkCaseFiles(projectId, agentId, benchmarkId, caseSummary.id, path, material)
.then((data) => {
if (cancelled) return;
setListing(data);
if (path === "" && !initialReadmeOpened.current) {
initialReadmeOpened.current = true;
const readme = data.entries.find(
(entry) => entry.kind === "file" && entry.name.toLowerCase() === "readme.md",
);
if (readme) onPreview(material, readme.name);
}
})
.catch((error: unknown) => {
if (!cancelled) setListError(apiErrorText(error));
});
return () => {
cancelled = true;
};
}, [projectId, agentId, benchmarkId, caseSummary.id, material, onPreview, open, path]);
const crumbs = path === "" ? [] : path.split("/");
const openEntry = (entry: WorkspaceFileEntry) => {
if (entry.kind === "dir") {
setPath(joinPath(path, entry.name));
return;
}
onPreview(material, joinPath(path, entry.name));
};
return (
<div className="border-b border-gray-200 last:border-b-0 dark:border-gray-800">
<button
type="button"
aria-expanded={open}
onClick={() => setOpen((value) => !value)}
className="flex w-full items-center gap-2 px-3 py-2 text-left hover:bg-gray-100 dark:hover:bg-gray-800/60"
>
<span className="text-xs text-gray-400">{open ? "▾" : "▸"}</span>
<span className="min-w-0 flex-1 text-sm font-medium">{label}</span>
{hiddenLabel && (
<span className="shrink-0 rounded bg-gray-200/70 px-1.5 py-0.5 text-[10px] text-gray-500 dark:bg-gray-800 dark:text-gray-400">
{hiddenLabel}
</span>
)}
</button>
{open && (
<div>
{crumbs.length > 0 && (
<div className="flex flex-wrap items-center gap-1 border-t border-gray-100 px-5 py-1.5 dark:border-gray-800/70">
<button
type="button"
onClick={() => setPath("")}
className="rounded px-1 py-0.5 text-xs text-gray-500 hover:bg-gray-100 dark:hover:bg-gray-800"
>
{label}
</button>
{crumbs.map((segment, index) => (
<span key={`${segment}-${index}`} className="flex min-w-0 items-center gap-1">
<span className="text-gray-300 dark:text-gray-700">/</span>
<button
type="button"
onClick={() => setPath(crumbs.slice(0, index + 1).join("/"))}
className="max-w-24 truncate rounded px-1 py-0.5 text-xs text-gray-500 hover:bg-gray-100 dark:hover:bg-gray-800"
>
{segment}
</button>
</span>
))}
</div>
)}
{listError && <p className="px-6 py-2 text-xs text-red-500">{listError}</p>}
{!listing && !listError && <SkeletonList rows={3} />}
{listing?.entries.length === 0 && (
<p className="px-6 py-2 text-xs text-gray-400">{S.files.empty}</p>
)}
{listing?.entries.map((entry) => (
<button
key={`${entry.kind}/${entry.name}`}
type="button"
onClick={() => openEntry(entry)}
className="flex w-full items-center gap-2 border-t border-gray-100 px-6 py-2 text-left hover:bg-gray-100 dark:border-gray-800/70 dark:hover:bg-gray-800/60"
>
<span className="text-sm text-gray-400">{entry.kind === "dir" ? "▸" : "·"}</span>
<span className="min-w-0 flex-1 truncate text-sm">{entry.name}</span>
{entry.kind === "file" && (
<span className="shrink-0 text-[11px] text-gray-400">
{formatBytes(entry.sizeBytes)}
</span>
)}
</button>
))}
</div>
)}
</div>
);
}
export function BenchmarkCaseBrowser({ projectId, agentId, benchmarkId, caseSummary }: Props) {
const [preview, setPreview] = useState<Preview | null>(null);
const previewRequest = useRef(0);
const fileUrl = useCallback(
(filePath: string, options?: { download?: boolean; preview?: boolean }) =>
api.benchmarkCaseFileUrl(projectId, agentId, benchmarkId, caseSummary.id, filePath, options),
(
material: CaseMaterial,
filePath: string,
options?: { download?: boolean; preview?: boolean },
) =>
api.benchmarkCaseFileUrl(
projectId,
agentId,
benchmarkId,
caseSummary.id,
filePath,
material,
options,
),
[projectId, agentId, benchmarkId, caseSummary.id],
);
const previewPath = useCallback(
async (filePath: string) => {
async (material: CaseMaterial, filePath: string) => {
const request = ++previewRequest.current;
const name = filePath.includes("/")
? filePath.slice(filePath.lastIndexOf("/") + 1)
: filePath;
const ext = extOf(name);
if (IMAGE_EXTS.has(ext)) {
setPreview({ path: filePath, name, kind: "image" });
setPreview({ material, path: filePath, name, kind: "image" });
return;
}
if (ext === "pdf") {
setPreview({ path: filePath, name, kind: "pdf" });
setPreview({ material, path: filePath, name, kind: "pdf" });
return;
}
const isMarkdown = ext === "md";
if (!TEXT_EXTS.has(ext)) {
setPreview({ path: filePath, name, kind: "unsupported" });
setPreview({ material, path: filePath, name, kind: "unsupported" });
return;
}
setPreview({ path: filePath, name, kind: isMarkdown ? "md" : "text", loading: true });
setPreview({
material,
path: filePath,
name,
kind: isMarkdown ? "md" : "text",
loading: true,
});
try {
const response = await fetch(fileUrl(filePath, { preview: true }), {
const response = await fetch(fileUrl(material, filePath, { preview: true }), {
credentials: "same-origin",
});
if (!response.ok) throw new Error(String(response.status));
const content = await response.text();
if (request !== previewRequest.current) return;
setPreview({
material,
path: filePath,
name,
kind: isMarkdown ? "md" : "text",
@@ -165,6 +311,7 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas
} catch (error) {
if (request !== previewRequest.current) return;
setPreview({
material,
path: filePath,
name,
kind: isMarkdown ? "md" : "text",
@@ -175,48 +322,19 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas
[fileUrl],
);
useEffect(() => {
setListing(null);
setListError(null);
let cancelled = false;
api
.listBenchmarkCaseFiles(projectId, agentId, benchmarkId, caseSummary.id, path)
.then((data) => {
if (cancelled) return;
setListing(data);
if (path === "" && !initialReadmeOpened.current) {
initialReadmeOpened.current = true;
const readme = data.entries.find(
(entry) => entry.kind === "file" && entry.name.toLowerCase() === "readme.md",
);
if (readme) void previewPath(readme.name);
}
})
.catch((error: unknown) => {
if (!cancelled) setListError(apiErrorText(error));
});
return () => {
cancelled = true;
};
}, [projectId, agentId, benchmarkId, caseSummary.id, path, previewPath]);
const crumbs = path === "" ? [] : path.split("/");
const downloadUrl = preview ? fileUrl(preview.path, { download: true }) : null;
const openEntry = (entry: WorkspaceFileEntry) => {
if (entry.kind === "dir") {
setPath(joinPath(path, entry.name));
return;
}
void previewPath(joinPath(path, entry.name));
};
const downloadUrl = preview ? fileUrl(preview.material, preview.path, { download: true }) : null;
const previewMaterialLabel =
preview?.material === "rubric" ? S.benchmark.rubric : S.benchmark.taskMaterials;
const markdownComponents: Components = {
img: ({ src, alt }) => (
<img
src={
typeof src === "string" && !EXTERNAL_REF_RE.test(src)
? fileUrl(resolveRelative(dirOf(preview?.path ?? ""), src))
? fileUrl(
preview?.material ?? "statement",
resolveRelative(dirOf(preview?.path ?? ""), src),
)
: src
}
alt={alt ?? ""}
@@ -234,12 +352,13 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas
);
}
const target = resolveRelative(dirOf(preview?.path ?? ""), href);
const material = preview?.material ?? "statement";
return (
<a
href={fileUrl(target)}
href={fileUrl(material, target)}
onClick={(event) => {
event.preventDefault();
void previewPath(target);
void previewPath(material, target);
}}
>
{children}
@@ -251,49 +370,27 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas
return (
<div className="grid min-h-[58vh] overflow-hidden rounded-md border border-gray-200 md:grid-cols-[240px_minmax(0,1fr)] dark:border-gray-800">
<aside className="border-b border-gray-200 bg-gray-50/60 md:border-b-0 md:border-r dark:border-gray-800 dark:bg-gray-950/30">
<div className="flex flex-wrap items-center gap-1 border-b border-gray-200 px-2 py-2 dark:border-gray-800">
<button
type="button"
onClick={() => setPath("")}
className="rounded px-1.5 py-0.5 text-xs text-gray-600 hover:bg-gray-100 dark:text-gray-300 dark:hover:bg-gray-800"
>
{S.benchmark.publicMaterials}
</button>
{crumbs.map((segment, index) => (
<span key={`${segment}-${index}`} className="flex min-w-0 items-center gap-1">
<span className="text-gray-300 dark:text-gray-700">/</span>
<button
type="button"
onClick={() => setPath(crumbs.slice(0, index + 1).join("/"))}
className="max-w-24 truncate rounded px-1 py-0.5 text-xs text-gray-600 hover:bg-gray-100 dark:text-gray-300 dark:hover:bg-gray-800"
>
{segment}
</button>
</span>
))}
</div>
<div className="max-h-44 overflow-y-auto md:max-h-[53vh]">
{listError && <p className="px-3 py-2 text-xs text-red-500">{listError}</p>}
{!listing && !listError && <SkeletonList rows={4} />}
{listing?.entries.length === 0 && (
<p className="px-3 py-2 text-xs text-gray-400">{S.files.empty}</p>
)}
{listing?.entries.map((entry) => (
<button
key={`${entry.kind}/${entry.name}`}
type="button"
onClick={() => openEntry(entry)}
className="flex w-full items-center gap-2 border-b border-gray-100 px-3 py-2 text-left hover:bg-gray-100 dark:border-gray-800/70 dark:hover:bg-gray-800/60"
>
<span className="text-sm text-gray-400">{entry.kind === "dir" ? "▸" : "·"}</span>
<span className="min-w-0 flex-1 truncate text-sm">{entry.name}</span>
{entry.kind === "file" && (
<span className="shrink-0 text-[11px] text-gray-400">
{formatBytes(entry.sizeBytes)}
</span>
)}
</button>
))}
<MaterialGroup
projectId={projectId}
agentId={agentId}
benchmarkId={benchmarkId}
caseSummary={caseSummary}
material="statement"
label={S.benchmark.taskMaterials}
defaultOpen
onPreview={previewPath}
/>
<MaterialGroup
projectId={projectId}
agentId={agentId}
benchmarkId={benchmarkId}
caseSummary={caseSummary}
material="rubric"
label={S.benchmark.rubric}
hiddenLabel={S.benchmark.agentHidden}
onPreview={previewPath}
/>
</div>
</aside>
@@ -301,7 +398,7 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas
<div className="flex min-h-11 flex-wrap items-center gap-2 border-b border-gray-200 px-3 py-2 dark:border-gray-800">
<div className="min-w-0 flex-1">
<p className="truncate font-mono text-xs text-gray-500">
{preview?.path ?? caseSummary.id}
{preview ? `${previewMaterialLabel} / ${preview.path}` : caseSummary.id}
</p>
</div>
{downloadUrl && preview && (
@@ -316,21 +413,21 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas
</div>
<div className="max-h-[52vh] min-h-[52vh] overflow-auto p-3">
{!preview ? (
<p className="text-sm text-gray-400">{S.benchmark.statementUnavailable}</p>
<p className="text-sm text-gray-400">{S.benchmark.caseFileUnavailable}</p>
) : preview.loading ? (
<SkeletonList rows={8} />
) : preview.error ? (
<p className="text-sm text-red-500">{preview.error}</p>
) : preview.kind === "image" ? (
<img
src={fileUrl(preview.path)}
src={fileUrl(preview.material, preview.path)}
alt={preview.name}
loading="lazy"
className="max-w-full rounded-md border border-gray-200 dark:border-gray-800"
/>
) : preview.kind === "pdf" ? (
<iframe
src={fileUrl(preview.path)}
src={fileUrl(preview.material, preview.path)}
title={preview.name}
className="h-[50vh] w-full rounded-md border border-gray-200 dark:border-gray-800"
/>
@@ -8,13 +8,50 @@ export interface MetricSourceLike {
score: number;
}
/** Each Evaluation's Score; non-finite malformed values become chart gaps defensively. */
/** Each Evaluation's authoritative stored Score; non-finite malformed values become chart gaps. */
export function scoreValues(evaluations: readonly MetricSourceLike[]): (number | null)[] {
return evaluations.map((e) => {
return typeof e.score === "number" && Number.isFinite(e.score) ? e.score : null;
});
}
export interface ScoreScale {
min: number;
max: number;
ticks: number[];
}
const SCORE_MIN = 0;
const SCORE_MAX = 100;
const SCORE_PADDING = 10;
const SCORE_TICK_STEPS = [1, 2, 2.5, 5, 10, 20];
/**
* Dynamic Score domain: pad the observed min/max by 10, clamp to the valid
* 0..100 Score interval, then round outward to human-friendly ticks.
*/
export function scoreScale(values: readonly (number | null)[]): ScoreScale {
const present = values.filter(
(value): value is number => value !== null && value >= SCORE_MIN && value <= SCORE_MAX,
);
if (present.length === 0) {
return { min: SCORE_MIN, max: SCORE_MAX, ticks: [0, 20, 40, 60, 80, 100] };
}
const observedMin = Math.min(...present);
const observedMax = Math.max(...present);
const paddedMin = Math.max(SCORE_MIN, observedMin - SCORE_PADDING);
const paddedMax = Math.min(SCORE_MAX, observedMax + SCORE_PADDING);
const step = SCORE_TICK_STEPS.find((candidate) => (paddedMax - paddedMin) / candidate <= 5)!;
const min = Math.max(SCORE_MIN, Math.floor(paddedMin / step) * step);
const max = Math.min(SCORE_MAX, Math.ceil(paddedMax / step) * step);
const ticks = Array.from(
{ length: Math.round((max - min) / step) + 1 },
(_, index) => min + index * step,
);
return { min, max, ticks };
}
/** A single data point on the chart: original index (x-axis position) + value. */
export interface MetricPoint {
index: number;
@@ -41,11 +78,6 @@ export function lineSegments(values: readonly (number | null)[]): MetricPoint[][
return segments;
}
/** Y-axis upper bound: the max of value-bearing points (falls back to a tiny positive number when all values are missing / zero, to avoid dividing by zero in the coordinate system). */
export function metricMax(values: readonly (number | null)[]): number {
return Math.max(1e-9, ...values.filter((v): v is number => v !== null));
}
/** Minimal evaluation shape needed for series grouping (BenchmarkEvaluation is a superset). */
export interface ModelRefLike {
modelId?: string;
@@ -31,17 +31,17 @@ import { EmptyState } from "../../components/ui/empty-state";
import { Modal } from "../../components/ui/modal";
import { SkeletonList } from "../../components/ui/skeleton";
import { seriesColor } from "../../lib/category-colors";
import { makeGeom } from "../usage/chart-geom";
import { makeRangeGeom } from "../usage/chart-geom";
import { ChartFrame, useChartWidth } from "../usage/chart-svg";
import {
lineSegments,
metricMax,
modelSeries,
scoreScale,
scoreValues,
seriesValues,
} from "./benchmark-metrics";
import type { EvaluationSeries } from "./benchmark-metrics";
import { BenchmarkStatementBrowser } from "./benchmark-statement-browser";
import { BenchmarkCaseBrowser } from "./benchmark-case-browser";
interface Selection {
agentId: string;
@@ -142,9 +142,9 @@ function AgentNode({
}
/**
* Score-over-time line chart. Every Run, Case, and Evaluation uses the fixed 0..100 scale.
* Evaluations remain grouped by model ID and thinking level so a runtime change stays visible
* without adding other metric modes.
* Score-over-time line chart. Scores remain valid on 0..100, while the visible y-axis is padded
* around the observed range and clamped to those limits. Evaluations remain grouped by model ID
* and thinking level so a runtime change stays visible without adding other metric modes.
*/
function ScoreTrendChart({
evaluations,
@@ -157,7 +157,8 @@ function ScoreTrendChart({
const [ref, width] = useChartWidth();
const values = scoreValues(evaluations);
const geom = makeGeom(evaluations.length, Math.max(metricMax(values), 100), width);
const scale = scoreScale(values);
const geom = makeRangeGeom(evaluations.length, scale.min, scale.max, width);
const dates = evaluations.map((e) => formatDateTime(e.time));
return (
@@ -169,6 +170,7 @@ function ScoreTrendChart({
dates={dates}
hover={hover}
onHover={setHover}
yTicks={scale.ticks}
bubble={(i) => {
const e = evaluations[i]!;
const v = values[i] ?? null;
@@ -650,7 +652,7 @@ export function BenchmarkPage() {
widthClass="sm:max-w-6xl"
onClose={() => setOpenCaseId(null)}
>
<BenchmarkStatementBrowser
<BenchmarkCaseBrowser
projectId={projectId}
agentId={selection.agentId}
benchmarkId={bm.id}
+18 -7
View File
@@ -28,9 +28,10 @@ export const PAD_B = 22;
/** The daily Token chart's three buckets (bottom-to-top stacking order is output → cacheWrite → cacheRead). */
export type TokenBucketKey = "cacheRead" | "cacheWrite" | "output";
/** A chart's coordinate system: canvas width w, data point count n, y-axis upper bound max, and x()/y() mapping "index / value" to canvas coordinates. */
/** A chart's coordinate system: canvas width w, data point count n, y-axis bounds, and x()/y() mapping "index / value" to canvas coordinates. */
export interface ChartGeom {
n: number;
min: number;
max: number;
/** Total canvas width (= viewBox width = CSS pixel width). */
w: number;
@@ -42,25 +43,35 @@ export interface ChartGeom {
}
/**
* Build the coordinate system: x takes each cell's midpoint, y runs
* top-to-bottom with max as the full height. w is the canvas width (pixels).
* When `max <= 0` (no data / all zero), y always takes the baseline —
* callers already guarantee max > 0, but this is an exported public pure
* function, and without this guard a single 0 would turn the entire chart's coordinates into NaN / Infinity.
* Build the zero-baseline coordinate system: x takes each cell's midpoint,
* y runs top-to-bottom with max as the full height, and w is the canvas width
* in pixels. Invalid or zero-height ranges map values to the baseline rather
* than producing NaN / Infinity.
*/
export function makeGeom(n: number, max: number, w: number): ChartGeom {
return makeRangeGeom(n, 0, max, w);
}
/**
* Build a coordinate system with an explicit y-axis range. Score charts use
* this to zoom into the observed values; zero-baseline usage charts keep
* calling makeGeom above.
*/
export function makeRangeGeom(n: number, min: number, max: number, w: number): ChartGeom {
const innerW = Math.max(0, w - PAD_L - PAD_R);
const innerH = CHART_H - PAD_T - PAD_B;
const step = n > 0 ? innerW / n : innerW;
const range = max - min;
return {
n,
min,
max,
w,
innerW,
innerH,
step,
x: (i) => PAD_L + step * i + step / 2,
y: (v) => PAD_T + innerH * (1 - (max > 0 ? v / max : 0)),
y: (v) => PAD_T + innerH * (1 - (range > 0 ? (v - min) / range : 0)),
};
}
@@ -76,6 +76,7 @@ export function ChartFrame({
bubble,
hitLayer,
labels,
yTicks,
hoverLine = true,
scrollToEnd = false,
children,
@@ -95,6 +96,8 @@ export function ChartFrame({
hitLayer?: ReactNode;
/** Indices for x-axis labels (omit for the default first/middle/last sparse labeling): the bar chart's cells are each wide, so it can label more via autoLabelIdx. */
labels?: number[];
/** Explicit y-axis ticks. Omit to divide the geom's min..max range into four equal intervals. */
yTicks?: number[];
/** Hover vertical indicator line (drawn by default): the bar chart turns it off — the bar itself already indicates the x position, so an extra line is just noise. */
hoverLine?: boolean;
/** Scroll to the far right by default when the canvas is wider than the container: the daily chart shows the most recent days first (scroll left for earlier ones). */
@@ -102,8 +105,9 @@ export function ChartFrame({
/** Data marks: bars / line / area, drawn between the grid and the hit area. */
children?: ReactNode;
}) {
const { x, y, w, innerH, step, max } = geom;
const gridLevels = [0, 0.25, 0.5, 0.75, 1].map((f) => max * f);
const { x, y, w, innerH, step, min, max } = geom;
const gridLevels =
yTicks ?? [0, 0.25, 0.5, 0.75, 1].map((fraction) => min + (max - min) * fraction);
const labelIdx = labels ?? sparseLabelIdx(dates.length);
const scrollRef = useRef<HTMLDivElement>(null);
+5 -3
View File
@@ -1016,9 +1016,11 @@ Scenarios:
caseCount: (n: number): string => `${n} case${n === 1 ? "" : "s"}`,
trendTitle: (metric: string): string => `${metric} over time`,
cases: "Cases",
viewCase: "View task",
publicMaterials: "Public materials",
statementUnavailable: "The public Statement is unavailable",
viewCase: "View details",
taskMaterials: "Task materials",
rubric: "Scoring rubric",
agentHidden: "Hidden from Target Agent",
caseFileUnavailable: "Case files are unavailable",
evaluations: "Evaluations",
noEvaluations: "No evaluations yet",
summaryLabel: "Summary",
+5 -3
View File
@@ -993,9 +993,11 @@ Benchmark:
/** Score-only chart title. */
trendTitle: (metric: string): string => `${metric}随时间变化`,
cases: "题目",
viewCase: "查看题目",
publicMaterials: "公开材料",
statementUnavailable: "题面暂时无法读取",
viewCase: "查看详情",
taskMaterials: "任务材料",
rubric: "评分标准",
agentHidden: "被测 Agent 不可见",
caseFileUnavailable: "案例文件暂时无法读取",
evaluations: "评估明细",
noEvaluations: "暂无评估记录",
/** Evaluation notes (scoreboard's summary: score source and notes on this round's changes). */
+44 -14
View File
@@ -1,12 +1,12 @@
/**
* Unit tests for the Evaluation center's Score-only chart helpers: Score extraction,
* gap segmentation across runtime series, y-axis max, and runtime grouping.
* dynamic y-axis range, gap segmentation across runtime series, and runtime grouping.
*/
import { describe, expect, it } from "vitest";
import {
lineSegments,
metricMax,
modelSeries,
scoreScale,
scoreValues,
seriesValues,
} from "../src/features/benchmark/benchmark-metrics";
@@ -14,12 +14,53 @@ import {
const evaluations = [{ score: 60 }, { score: 75.25 }, { score: 85.5 }];
describe("scoreValues", () => {
it("extracts stored Scores and treats non-finite malformed input as a gap", () => {
it("extracts stored Scores and treats non-finite input as a gap", () => {
expect(scoreValues(evaluations)).toEqual([60, 75.25, 85.5]);
expect(scoreValues([{ score: Number.NaN }, { score: Infinity }])).toEqual([null, null]);
});
});
describe("scoreScale (dynamic padded Score axis)", () => {
it("pads observed scores, clamps to 0..100, and rounds outward to friendly ticks", () => {
expect(scoreScale([71, 83.67, 88.33])).toEqual({
min: 60,
max: 100,
ticks: [60, 70, 80, 90, 100],
});
});
it("keeps a dynamic range for a single or repeated score", () => {
expect(scoreScale([88])).toEqual({
min: 75,
max: 100,
ticks: [75, 80, 85, 90, 95, 100],
});
expect(scoreScale([50, 50])).toEqual({
min: 40,
max: 60,
ticks: [40, 45, 50, 55, 60],
});
});
it("clamps boundary scores and falls back safely when every value is missing", () => {
expect(scoreScale([100])).toEqual({
min: 90,
max: 100,
ticks: [90, 92, 94, 96, 98, 100],
});
expect(scoreScale([0])).toEqual({
min: 0,
max: 10,
ticks: [0, 2, 4, 6, 8, 10],
});
expect(scoreScale([null, null])).toEqual({
min: 0,
max: 100,
ticks: [0, 20, 40, 60, 80, 100],
});
});
});
describe("lineSegments (gap segmentation)", () => {
it("no gaps: one segment with everything (consecutive indexes)", () => {
expect(lineSegments([60, 75.25, 85.5])).toEqual([
@@ -51,17 +92,6 @@ describe("lineSegments (gap segmentation)", () => {
});
});
describe("metricMax (y-axis upper bound)", () => {
it("takes the maximum of present points (ignoring null)", () => {
expect(metricMax([0.12, null, 0.2])).toBe(0.2);
});
it("all missing / all zero yields a tiny positive number (no division by zero in the coordinate system)", () => {
expect(metricMax([null, null])).toBe(1e-9);
expect(metricMax([0, 0])).toBe(1e-9);
});
});
describe("modelSeries / seriesValues (curves split by model ID and thinking level)", () => {
const mixed = [
{
+10
View File
@@ -15,6 +15,7 @@
import { describe, expect, it } from "vitest";
import {
makeGeom,
makeRangeGeom,
linePath,
areaPath,
sparseLabelIdx,
@@ -56,6 +57,15 @@ describe("makeGeom", () => {
it("step falls back to the whole inner width when n=0 (no division by zero)", () => {
expect(makeGeom(0, 1, 640).step).toBe(586);
});
it("an explicit non-zero range maps its min to the baseline and max to the top", () => {
const g = makeRangeGeom(2, 60, 100, 640);
expect(g.min).toBe(60);
expect(g.max).toBe(100);
expect(g.y(60)).toBe(178);
expect(g.y(100)).toBe(10);
expect(g.y(80)).toBe(94);
});
});
describe("linePath / areaPath", () => {