diff --git a/changelog/unreleased/2026-07-31-evaluation-center-case-details.md b/changelog/unreleased/2026-07-31-evaluation-center-case-details.md new file mode 100644 index 0000000..ccbf46e --- /dev/null +++ b/changelog/unreleased/2026-07-31-evaluation-center-case-details.md @@ -0,0 +1,9 @@ +# Evaluation Center Case details and Score charts + +Case details now separate the task materials visible to the Target Agent from the scoring rubric available to project members, while keeping both file roots path-confined. Score charts use a padded dynamic axis without discarding authoritative stored values, and the benchmark Skills use YAML folded scalars for Scoreboard summaries. + +## Details + +- Case details show task materials and scoring rubrics as separate file groups. Rubrics remain hidden from the Target Agent. +- Score charts pad the observed in-range values, clamp the displayed axis to 0–100, and preserve stored finite values for plotting and tooltips. +- Benchmark design and optimization Skills write summary fields as YAML folded scalars and parse the complete Scoreboard after appending an Evaluation. diff --git a/changelog/unreleased/README.md b/changelog/unreleased/README.md index 1a34750..5370843 100644 --- a/changelog/unreleased/README.md +++ b/changelog/unreleased/README.md @@ -2,4 +2,6 @@ Changes since v0.1.5. The version number is assigned at release, when this folder is renamed. +- [2026-07-31] Evaluation Center: Case details separate Target Agent task materials from project-member-visible scoring rubrics, Score charts use a padded dynamic axis without discarding stored values, and the benchmark Skills write YAML-safe Scoreboard summaries. ([details](2026-07-31-evaluation-center-case-details.md)) + - [2026-07-30] Web App: the main conversation's file summary moves to the completed Task boundary — one card per Task scanning all of its assistant text, nested agent conversations keep their per-message summaries, and file-existence caching stops retaining negative results so a later Task can surface a newly created path. ([details](2026-07-30-file-summary-task-boundary.md)) diff --git a/packages/server/src/api/types.ts b/packages/server/src/api/types.ts index cd62f04..d6d207e 100644 --- a/packages/server/src/api/types.ts +++ b/packages/server/src/api/types.ts @@ -1292,6 +1292,8 @@ export interface BenchmarksResponse { benchmarks: BenchmarkSummary[]; } +export type CaseMaterial = "statement" | "rubric"; + /** Public Benchmark Case metadata. Rubric and Gold content are never included. */ export interface BenchmarkCaseSummary { id: string; diff --git a/packages/server/src/http/routes/benchmarks.ts b/packages/server/src/http/routes/benchmarks.ts index 51f1f84..ac7923c 100644 --- a/packages/server/src/http/routes/benchmarks.ts +++ b/packages/server/src/http/routes/benchmarks.ts @@ -4,16 +4,72 @@ * GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases * GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/files * GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/files/content + * GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/rubric/files + * GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/rubric/files/content * Returns the Agent's Benchmark list (title/description from benchmark_config.toml) * along with the evaluations[] from scoreboard.yaml. */ -import { Hono } from "hono"; +import { Hono, type Context } from "hono"; import type { AppEnv } from "../../auth/middleware.js"; import type { AppDeps } from "../../app.js"; +import type { CaseMaterial } from "../../api/types.js"; import { requireValidId } from "../validate.js"; const TEXT_PREVIEW_BYTES = 256 * 1024; +function listCaseFiles(deps: AppDeps, material: CaseMaterial) { + return async (c: Context) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + const benchmarkId = requireValidId(c, "benchmarkId"); + const caseId = requireValidId(c, "caseId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + return c.json( + await deps.benchmarks.listCaseFiles( + projectId, + agentId, + benchmarkId, + caseId, + c.req.query("path") ?? "", + material, + ), + ); + }; +} + +function readCaseFile(deps: AppDeps, material: CaseMaterial) { + return async (c: Context) => { + const projectId = requireValidId(c, "projectId"); + const agentId = requireValidId(c, "agentId"); + const benchmarkId = requireValidId(c, "benchmarkId"); + const caseId = requireValidId(c, "caseId"); + deps.projectService.requireProjectAccess(c.var.user.userId, projectId); + await deps.agentConfigService.requireExists(projectId, agentId); + const download = c.req.query("download") === "1"; + const boundedPreview = !download && c.req.query("preview") === "1"; + const { data, fileName, contentType, scriptable, truncated } = + await deps.benchmarks.readCaseFile( + projectId, + agentId, + benchmarkId, + caseId, + c.req.query("path") ?? "", + material, + boundedPreview ? { maxBytes: TEXT_PREVIEW_BYTES } : undefined, + ); + return new Response(new Uint8Array(data), { + status: 200, + headers: { + "Content-Type": !download && scriptable ? "text/plain; charset=utf-8" : contentType, + "Content-Disposition": `${download ? "attachment" : "inline"}; filename*=UTF-8''${encodeURIComponent(fileName)}`, + "X-Content-Type-Options": "nosniff", + ...(truncated ? { "X-Content-Truncated": "1" } : {}), + }, + }); + }; +} + export function benchmarksRoutes(deps: AppDeps): Hono { const app = new Hono(); @@ -34,52 +90,10 @@ export function benchmarksRoutes(deps: AppDeps): Hono { return c.json(await deps.benchmarks.listCases(projectId, agentId, benchmarkId)); }); - app.get("/:benchmarkId/cases/:caseId/files", async (c) => { - const projectId = requireValidId(c, "projectId"); - const agentId = requireValidId(c, "agentId"); - const benchmarkId = requireValidId(c, "benchmarkId"); - const caseId = requireValidId(c, "caseId"); - deps.projectService.requireProjectAccess(c.var.user.userId, projectId); - await deps.agentConfigService.requireExists(projectId, agentId); - return c.json( - await deps.benchmarks.listCaseFiles( - projectId, - agentId, - benchmarkId, - caseId, - c.req.query("path") ?? "", - ), - ); - }); - - app.get("/:benchmarkId/cases/:caseId/files/content", async (c) => { - const projectId = requireValidId(c, "projectId"); - const agentId = requireValidId(c, "agentId"); - const benchmarkId = requireValidId(c, "benchmarkId"); - const caseId = requireValidId(c, "caseId"); - deps.projectService.requireProjectAccess(c.var.user.userId, projectId); - await deps.agentConfigService.requireExists(projectId, agentId); - const download = c.req.query("download") === "1"; - const boundedPreview = !download && c.req.query("preview") === "1"; - const { data, fileName, contentType, scriptable, truncated } = - await deps.benchmarks.readCaseFile( - projectId, - agentId, - benchmarkId, - caseId, - c.req.query("path") ?? "", - boundedPreview ? { maxBytes: TEXT_PREVIEW_BYTES } : undefined, - ); - return new Response(new Uint8Array(data), { - status: 200, - headers: { - "Content-Type": !download && scriptable ? "text/plain; charset=utf-8" : contentType, - "Content-Disposition": `${download ? "attachment" : "inline"}; filename*=UTF-8''${encodeURIComponent(fileName)}`, - "X-Content-Type-Options": "nosniff", - ...(truncated ? { "X-Content-Truncated": "1" } : {}), - }, - }); - }); + app.get("/:benchmarkId/cases/:caseId/files", listCaseFiles(deps, "statement")); + app.get("/:benchmarkId/cases/:caseId/files/content", readCaseFile(deps, "statement")); + app.get("/:benchmarkId/cases/:caseId/rubric/files", listCaseFiles(deps, "rubric")); + app.get("/:benchmarkId/cases/:caseId/rubric/files/content", readCaseFile(deps, "rubric")); return app; } diff --git a/packages/server/src/services/benchmark-service.ts b/packages/server/src/services/benchmark-service.ts index 84bdb8e..28c7cbe 100644 --- a/packages/server/src/services/benchmark-service.ts +++ b/packages/server/src/services/benchmark-service.ts @@ -25,6 +25,7 @@ import type { BenchmarkRunScore, BenchmarkSummary, BenchmarksResponse, + CaseMaterial, WorkspaceFilesResponse, } from "../api/types.js"; import type { @@ -235,7 +236,13 @@ export class BenchmarkService { .sort((a, b) => a.name.localeCompare(b.name))) { const fallback: BenchmarkCaseSummary = { id: entry.name, title: entry.name }; try { - const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, entry.name); + const statementDir = await this.caseMaterialRoot( + projectId, + agentId, + benchmarkId, + entry.name, + "statement", + ); const realReadme = await fs.realpath(path.join(statementDir, "README.md")); if (!isWithin(statementDir, realReadme)) throw new Error("README escapes Statement"); cases.push({ @@ -255,9 +262,16 @@ export class BenchmarkService { benchmarkId: string, caseId: string, rel: string, + material: CaseMaterial, ): Promise { - const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, caseId); - return this.workspaceFiles.list(statementDir, rel); + const materialRoot = await this.caseMaterialRoot( + projectId, + agentId, + benchmarkId, + caseId, + material, + ); + return this.workspaceFiles.list(materialRoot, rel); } async readCaseFile( @@ -266,37 +280,45 @@ export class BenchmarkService { benchmarkId: string, caseId: string, rel: string, + material: CaseMaterial, options?: WorkspaceFileReadOptions, ): Promise { - const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, caseId); - return this.workspaceFiles.read(statementDir, rel, options); + const materialRoot = await this.caseMaterialRoot( + projectId, + agentId, + benchmarkId, + caseId, + material, + ); + return this.workspaceFiles.read(materialRoot, rel, options); } - private async statementRoot( + private async caseMaterialRoot( projectId: string, agentId: string, benchmarkId: string, caseId: string, + material: CaseMaterial, ): Promise { const benchDir = path.join(benchmarksDir(this.root, projectId, agentId), benchmarkId); const caseDir = path.join(benchDir, caseId); - const statementDir = path.join(caseDir, "statement"); + const materialRoot = path.join(caseDir, material); try { - const [realBenchDir, realCaseDir, realStatementDir] = await Promise.all([ + const [realBenchDir, realCaseDir, realMaterialRoot] = await Promise.all([ fs.realpath(benchDir), fs.realpath(caseDir), - fs.realpath(statementDir), + fs.realpath(materialRoot), ]); if ( !isWithin(realBenchDir, realCaseDir) || - path.dirname(realStatementDir) !== realCaseDir || - path.basename(realStatementDir) !== "statement" + path.dirname(realMaterialRoot) !== realCaseDir || + path.basename(realMaterialRoot) !== material ) { - throw new Error("Statement path is not canonical"); + throw new Error("Case material is not canonical"); } - return realStatementDir; + return realMaterialRoot; } catch { - throw new HttpError(404, "not_found", "Public Case Statement does not exist."); + throw new HttpError(404, "not_found", `Case ${material} does not exist.`); } } diff --git a/packages/server/test/benchmarks.test.ts b/packages/server/test/benchmarks.test.ts index e0dc56a..7ee2385 100644 --- a/packages/server/test/benchmarks.test.ts +++ b/packages/server/test/benchmarks.test.ts @@ -69,6 +69,7 @@ describe("benchmarks api", () => { await fs.mkdir(path.join(dir, "CASE-001-excel-task", "statement", "assets"), { recursive: true, }); + await fs.mkdir(path.join(dir, "CASE-001-excel-task", "rubric"), { recursive: true }); await fs.mkdir(path.join(dir, "CASE-002-web-task", "statement"), { recursive: true }); await fs.mkdir(path.join(dir, "CASE-002-web-task", "rubric"), { recursive: true }); await fs.writeFile( @@ -91,6 +92,16 @@ describe("benchmarks api", () => { "x".repeat(300 * 1024), "utf8", ); + await fs.writeFile( + path.join(dir, "CASE-001-excel-task", "rubric", "README.md"), + "# Scoring rubric\n\n- Correct workbook: 100 points", + "utf8", + ); + await fs.writeFile( + path.join(dir, "CASE-001-excel-task", "rubric", "expected.json"), + '{"rows": 1}\n', + "utf8", + ); await fs.writeFile( path.join(dir, "CASE-002-web-task", "statement", "README.md"), "# Web task\n\nBuild the page.", @@ -245,6 +256,22 @@ describe("benchmarks api", () => { expect(statement.headers.get("content-type")).toContain("markdown"); expect(await statement.text()).toBe("# Case 001: Excel cleanup\n\nClean the workbook."); + const rubricFilesBase = `${base}/swe-bench-v2/cases/CASE-001-excel-task/rubric/files`; + const rubricFiles = (await ( + await member.get(rubricFilesBase) + ).json()) as WorkspaceFilesResponse; + expect(rubricFiles.entries.map((entry) => `${entry.kind}:${entry.name}`)).toEqual([ + "file:expected.json", + "file:README.md", + ]); + + const rubric = await member.get( + `${rubricFilesBase}/content?path=${encodeURIComponent("README.md")}&preview=1`, + ); + expect(rubric.status).toBe(200); + expect(rubric.headers.get("content-type")).toContain("markdown"); + expect(await rubric.text()).toContain("Correct workbook: 100 points"); + const large = await member.get( `${filesBase}/content?path=${encodeURIComponent("large.txt")}&preview=1`, ); @@ -262,6 +289,13 @@ describe("benchmarks api", () => { (await member.get(`${filesBase}/content?path=${encodeURIComponent("../rubric/README.md")}`)) .status, ).toBe(400); + expect( + ( + await member.get( + `${rubricFilesBase}/content?path=${encodeURIComponent("../statement/README.md")}`, + ) + ).status, + ).toBe(400); expect((await outsider.get(`${base}/swe-bench-v2/cases`)).status).toBe(404); expect((await outsider.get(filesBase)).status).toBe(404); }); diff --git a/packages/skills/skills/agent-optimization/SKILL.md b/packages/skills/skills/agent-optimization/SKILL.md index 284fbc4..565f4f4 100644 --- a/packages/skills/skills/agent-optimization/SKILL.md +++ b/packages/skills/skills/agent-optimization/SKILL.md @@ -3,8 +3,8 @@ name: agent-optimization description: Improve an Agent State through versioned scores and score-linked Traces from a frozen Benchmark. short_description: Improve an Agent from measured Benchmark results. short_description_zh: 根据 Benchmark 结果改进 Agent。 -version: 7 -updated: 2026-07-30T02:51:10Z +version: 8 +updated: 2026-07-31T04:10:53Z --- # Agent Optimization @@ -106,8 +106,10 @@ Append each complete accepted Candidate Evaluation to `scoreboard.yaml` immediat provider: model_id: thinking_level: - summary_title: - summary: + summary_title: >- + + summary: >- + score: cost: duration_ms: @@ -123,6 +125,8 @@ Append each complete accepted Candidate Evaluation to `scoreboard.yaml` immediat session_id: ``` +After writing, parse the complete `scoreboard.yaml` and verify the appended Evaluation before reporting success or continuing. + Every Run and Case score is on the fixed `0..100` scale. Do not write `max_score`. Calculate and write every Case and Evaluation average directly in the Scoreboard: ignore `null` values when averaging cost and write `null` only when all contributing costs are unknown; round `score` averages to two decimal places, `cost` averages to six decimal places, and `duration_ms` averages to the nearest integer. These stored values are authoritative—do not add a server, frontend, script, or consistency check that recomputes or validates them. Do not add an `aggregate` object or use `case_id`, `mean_score`, `mean_cost`, or `mean_duration_ms`. Do not record rejected Candidates in the Scoreboard. Report the Baseline and every fully evaluated Candidate with its score, version, change, decision, and Test Session ids. For each Candidate, distinguish the acceptance decision from whether its stated hypothesis was supported by the predicted Case behavior. Include the final retained version, stop reason, and known limitations. Never report a score for an Agent State that was not evaluated. diff --git a/packages/skills/skills/benchmark-design/SKILL.md b/packages/skills/skills/benchmark-design/SKILL.md index fcc267e..8bb0741 100644 --- a/packages/skills/skills/benchmark-design/SKILL.md +++ b/packages/skills/skills/benchmark-design/SKILL.md @@ -3,8 +3,8 @@ name: benchmark-design description: Design and calibrate a multi-Case capability Benchmark and establish a traceable Formal Baseline. short_description: Design and calibrate an Agent capability Benchmark. short_description_zh: 设计并校准 Agent 能力评测 Benchmark。 -version: 5 -updated: 2026-07-29T17:20:58Z +version: 6 +updated: 2026-07-31T04:10:53Z --- # Benchmark Design @@ -175,8 +175,10 @@ evaluations: provider: model_id: thinking_level: - summary_title: - summary: + summary_title: >- + + summary: >- + score: cost: duration_ms: @@ -192,6 +194,8 @@ evaluations: session_id: ``` +After writing, parse the complete `scoreboard.yaml` and verify the appended Evaluation before reporting success or continuing. + Every Run and Case score is on the fixed `0..100` scale. Do not write `max_score`. Calculate and write every Case and Evaluation average directly in the Scoreboard: ignore `null` values when averaging cost and write `null` only when all contributing costs are unknown; round `score` averages to two decimal places, `cost` averages to six decimal places, and `duration_ms` averages to the nearest integer. These stored values are authoritative—do not add a server, frontend, script, or consistency check that recomputes or validates them. Do not add an `aggregate` object or use `case_id`, `mean_score`, `mean_cost`, or `mean_duration_ms`. Report the Benchmark path, configuration, Agent State version, Evaluation average and Case Run scores, Test Session ids, and known limitations. Include one compact row per Pilot iteration with its score, diagnosed capability gap, difficulty adjustment, and freeze or stop decision. diff --git a/packages/skills/test/skills.test.ts b/packages/skills/test/skills.test.ts index cd540c6..e3fbed6 100644 --- a/packages/skills/test/skills.test.ts +++ b/packages/skills/test/skills.test.ts @@ -300,6 +300,11 @@ describe("librarySkill", () => { expect(benchmarkDesign).toContain("ignore `null` values when averaging cost"); expect(benchmarkDesign).toContain("stored values are authoritative"); expect(benchmarkDesign).toContain("obtain the current UTC timestamp from the environment"); + expect(benchmarkDesignRaw).toMatch(/summary_title:\s*>-\n\s+/); + expect(benchmarkDesignRaw).toMatch(/summary:\s*>-\n\s+/); + expect(benchmarkDesign).toContain( + "parse the complete `scoreboard.yaml` and verify the appended Evaluation", + ); expect(benchmarkDesignRaw).not.toMatch(/\n\s+max_score:/); expect(benchmarkDesign).toContain( "Before reading `status`, `score`, or any other protocol field", @@ -308,6 +313,11 @@ describe("librarySkill", () => { expect(optimization).toContain("Delegate every evaluation to an `agent-evaluation` subagent"); expect(optimization).toContain("Before reading `status`, `score`, or any other protocol field"); expect(optimization).toContain("Ask the same Evaluator to resend only the clean YAML"); + expect(optimizationRaw).toMatch(/summary_title:\s*>-\n\s+/); + expect(optimizationRaw).toMatch(/summary:\s*>-\n\s+/); + expect(optimization).toContain( + "parse the complete `scoreboard.yaml` and verify the appended Evaluation", + ); expect(optimization).toContain( "A round counts only after one Candidate has a complete valid Evaluation", ); diff --git a/packages/web/src/api/endpoints.ts b/packages/web/src/api/endpoints.ts index b7e22d7..3872972 100644 --- a/packages/web/src/api/endpoints.ts +++ b/packages/web/src/api/endpoints.ts @@ -23,6 +23,7 @@ import type { AuthResponse, BenchmarkCasesResponse, BenchmarksResponse, + CaseMaterial, DirListResponse, FilesStatRequest, FilesStatResponse, @@ -538,16 +539,27 @@ export const listBenchmarkCases = (projectId: string, agentId: string, benchmark `/benchmarks/${encodeURIComponent(benchmarkId)}/cases`, ); +const benchmarkCaseFilesPath = ( + projectId: string, + agentId: string, + benchmarkId: string, + caseId: string, + material: CaseMaterial, +) => + `/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` + + `/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}` + + `${material === "rubric" ? "/rubric" : ""}/files`; + export const listBenchmarkCaseFiles = ( projectId: string, agentId: string, benchmarkId: string, caseId: string, path: string, + material: CaseMaterial, ) => apiFetch( - `/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` + - `/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}/files`, + benchmarkCaseFilesPath(projectId, agentId, benchmarkId, caseId, material), { query: { path } }, ); @@ -557,12 +569,16 @@ export const benchmarkCaseFileUrl = ( benchmarkId: string, caseId: string, path: string, + material: CaseMaterial, options?: { download?: boolean; preview?: boolean }, ): string => { - const base = - `/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` + - `/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}` + - `/files/content?path=${encodeURIComponent(path)}`; + const base = `${benchmarkCaseFilesPath( + projectId, + agentId, + benchmarkId, + caseId, + material, + )}/content?path=${encodeURIComponent(path)}`; return ( base + (options?.download ? "&download=1" : "") + diff --git a/packages/web/src/features/benchmark/benchmark-statement-browser.tsx b/packages/web/src/features/benchmark/benchmark-case-browser.tsx similarity index 64% rename from packages/web/src/features/benchmark/benchmark-statement-browser.tsx rename to packages/web/src/features/benchmark/benchmark-case-browser.tsx index 45e73ec..8b8fc7b 100644 --- a/packages/web/src/features/benchmark/benchmark-statement-browser.tsx +++ b/packages/web/src/features/benchmark/benchmark-case-browser.tsx @@ -1,6 +1,7 @@ import { useCallback, useEffect, useRef, useState } from "react"; import type { BenchmarkCaseSummary, + CaseMaterial, WorkspaceFileEntry, WorkspaceFilesResponse, } from "@prismshadow/penguin-server/api"; @@ -46,6 +47,7 @@ const EXTERNAL_REF_RE = /^[a-z][a-z0-9+.-]*:/i; const HIGHLIGHT_LIMIT = 64 * 1024; interface Preview { + material: CaseMaterial; path: string; name: string; kind: "text" | "md" | "image" | "pdf" | "unsupported"; @@ -62,6 +64,14 @@ interface Props { caseSummary: BenchmarkCaseSummary; } +interface MaterialGroupProps extends Props { + material: CaseMaterial; + label: string; + hiddenLabel?: string; + defaultOpen?: boolean; + onPreview: (material: CaseMaterial, path: string) => void; +} + function extOf(name: string): string { const index = name.lastIndexOf("."); return index >= 0 ? name.slice(index + 1).toLowerCase() : name.toLowerCase(); @@ -113,49 +123,185 @@ function languageFor(name: string): string { ); } -export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, caseSummary }: Props) { +function MaterialGroup({ + projectId, + agentId, + benchmarkId, + caseSummary, + material, + label, + hiddenLabel, + defaultOpen = false, + onPreview, +}: MaterialGroupProps) { + const [open, setOpen] = useState(defaultOpen); const [path, setPath] = useState(""); const [listing, setListing] = useState(null); const [listError, setListError] = useState(null); - const [preview, setPreview] = useState(null); const initialReadmeOpened = useRef(false); + + useEffect(() => { + if (!open) return; + setListing(null); + setListError(null); + let cancelled = false; + api + .listBenchmarkCaseFiles(projectId, agentId, benchmarkId, caseSummary.id, path, material) + .then((data) => { + if (cancelled) return; + setListing(data); + if (path === "" && !initialReadmeOpened.current) { + initialReadmeOpened.current = true; + const readme = data.entries.find( + (entry) => entry.kind === "file" && entry.name.toLowerCase() === "readme.md", + ); + if (readme) onPreview(material, readme.name); + } + }) + .catch((error: unknown) => { + if (!cancelled) setListError(apiErrorText(error)); + }); + return () => { + cancelled = true; + }; + }, [projectId, agentId, benchmarkId, caseSummary.id, material, onPreview, open, path]); + + const crumbs = path === "" ? [] : path.split("/"); + + const openEntry = (entry: WorkspaceFileEntry) => { + if (entry.kind === "dir") { + setPath(joinPath(path, entry.name)); + return; + } + onPreview(material, joinPath(path, entry.name)); + }; + + return ( +
+ + {open && ( +
+ {crumbs.length > 0 && ( +
+ + {crumbs.map((segment, index) => ( + + / + + + ))} +
+ )} + {listError &&

{listError}

} + {!listing && !listError && } + {listing?.entries.length === 0 && ( +

{S.files.empty}

+ )} + {listing?.entries.map((entry) => ( + + ))} +
+ )} +
+ ); +} + +export function BenchmarkCaseBrowser({ projectId, agentId, benchmarkId, caseSummary }: Props) { + const [preview, setPreview] = useState(null); const previewRequest = useRef(0); const fileUrl = useCallback( - (filePath: string, options?: { download?: boolean; preview?: boolean }) => - api.benchmarkCaseFileUrl(projectId, agentId, benchmarkId, caseSummary.id, filePath, options), + ( + material: CaseMaterial, + filePath: string, + options?: { download?: boolean; preview?: boolean }, + ) => + api.benchmarkCaseFileUrl( + projectId, + agentId, + benchmarkId, + caseSummary.id, + filePath, + material, + options, + ), [projectId, agentId, benchmarkId, caseSummary.id], ); const previewPath = useCallback( - async (filePath: string) => { + async (material: CaseMaterial, filePath: string) => { const request = ++previewRequest.current; const name = filePath.includes("/") ? filePath.slice(filePath.lastIndexOf("/") + 1) : filePath; const ext = extOf(name); if (IMAGE_EXTS.has(ext)) { - setPreview({ path: filePath, name, kind: "image" }); + setPreview({ material, path: filePath, name, kind: "image" }); return; } if (ext === "pdf") { - setPreview({ path: filePath, name, kind: "pdf" }); + setPreview({ material, path: filePath, name, kind: "pdf" }); return; } const isMarkdown = ext === "md"; if (!TEXT_EXTS.has(ext)) { - setPreview({ path: filePath, name, kind: "unsupported" }); + setPreview({ material, path: filePath, name, kind: "unsupported" }); return; } - setPreview({ path: filePath, name, kind: isMarkdown ? "md" : "text", loading: true }); + setPreview({ + material, + path: filePath, + name, + kind: isMarkdown ? "md" : "text", + loading: true, + }); try { - const response = await fetch(fileUrl(filePath, { preview: true }), { + const response = await fetch(fileUrl(material, filePath, { preview: true }), { credentials: "same-origin", }); if (!response.ok) throw new Error(String(response.status)); const content = await response.text(); if (request !== previewRequest.current) return; setPreview({ + material, path: filePath, name, kind: isMarkdown ? "md" : "text", @@ -165,6 +311,7 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas } catch (error) { if (request !== previewRequest.current) return; setPreview({ + material, path: filePath, name, kind: isMarkdown ? "md" : "text", @@ -175,48 +322,19 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas [fileUrl], ); - useEffect(() => { - setListing(null); - setListError(null); - let cancelled = false; - api - .listBenchmarkCaseFiles(projectId, agentId, benchmarkId, caseSummary.id, path) - .then((data) => { - if (cancelled) return; - setListing(data); - if (path === "" && !initialReadmeOpened.current) { - initialReadmeOpened.current = true; - const readme = data.entries.find( - (entry) => entry.kind === "file" && entry.name.toLowerCase() === "readme.md", - ); - if (readme) void previewPath(readme.name); - } - }) - .catch((error: unknown) => { - if (!cancelled) setListError(apiErrorText(error)); - }); - return () => { - cancelled = true; - }; - }, [projectId, agentId, benchmarkId, caseSummary.id, path, previewPath]); - - const crumbs = path === "" ? [] : path.split("/"); - const downloadUrl = preview ? fileUrl(preview.path, { download: true }) : null; - - const openEntry = (entry: WorkspaceFileEntry) => { - if (entry.kind === "dir") { - setPath(joinPath(path, entry.name)); - return; - } - void previewPath(joinPath(path, entry.name)); - }; + const downloadUrl = preview ? fileUrl(preview.material, preview.path, { download: true }) : null; + const previewMaterialLabel = + preview?.material === "rubric" ? S.benchmark.rubric : S.benchmark.taskMaterials; const markdownComponents: Components = { img: ({ src, alt }) => ( {alt { event.preventDefault(); - void previewPath(target); + void previewPath(material, target); }} > {children} @@ -251,49 +370,27 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas return (
@@ -301,7 +398,7 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas

- {preview?.path ?? caseSummary.id} + {preview ? `${previewMaterialLabel} / ${preview.path}` : caseSummary.id}

{downloadUrl && preview && ( @@ -316,21 +413,21 @@ export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, cas
{!preview ? ( -

{S.benchmark.statementUnavailable}

+

{S.benchmark.caseFileUnavailable}

) : preview.loading ? ( ) : preview.error ? (

{preview.error}

) : preview.kind === "image" ? ( {preview.name} ) : preview.kind === "pdf" ? (