feat: 完善两阶段 Agent 自进化 Pipeline、Skills 与 Benchmark (#129)

This commit is contained in:
Jingzhe Xu
2026-07-30 15:59:58 +08:00
committed by GitHub
parent 3bd0a7208a
commit 081fe2d172
34 changed files with 2093 additions and 763 deletions
+39
View File
@@ -21,6 +21,7 @@ import type {
ApprovalDecisionRequest,
AuthLoginRequest,
AuthResponse,
BenchmarkCasesResponse,
BenchmarksResponse,
DirListResponse,
FilesStatRequest,
@@ -522,6 +523,44 @@ export const listBenchmarks = (projectId: string, agentId: string) =>
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}/benchmarks`,
);
export const listBenchmarkCases = (projectId: string, agentId: string, benchmarkId: string) =>
apiFetch<BenchmarkCasesResponse>(
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` +
`/benchmarks/${encodeURIComponent(benchmarkId)}/cases`,
);
export const listBenchmarkCaseFiles = (
projectId: string,
agentId: string,
benchmarkId: string,
caseId: string,
path: string,
) =>
apiFetch<WorkspaceFilesResponse>(
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` +
`/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}/files`,
{ query: { path } },
);
export const benchmarkCaseFileUrl = (
projectId: string,
agentId: string,
benchmarkId: string,
caseId: string,
path: string,
options?: { download?: boolean; preview?: boolean },
): string => {
const base =
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` +
`/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}` +
`/files/content?path=${encodeURIComponent(path)}`;
return (
base +
(options?.download ? "&download=1" : "") +
(options?.preview && !options.download ? "&preview=1" : "")
);
};
// Agent State snapshot export / import ------------------------------------------------------
/** Snapshot bundle (tar.gz) download URL: the server sets Content-Disposition attachment, usable directly in <a download>. */
@@ -1,32 +1,17 @@
/**
* Metric switching and per-model series grouping for the Benchmark center chart (pure functions,
* easy to unit test): the chart can switch between the score / cost / duration metrics (sharing
* the same time axis), and is grouped into series by each evaluation's (provider, modelId) — one
* color per series, legend by model. score is always present;
* cost / durationMs are optional — missing values are **skipped points**: neither drawn nor
* connected, so the line breaks at the gap (lineSegments splits value-bearing indices into
* contiguous segments).
* Score-only data helpers and per-runtime series grouping for the Benchmark center chart.
* Series share one time axis and use each Evaluation's authoritative stored Score.
*/
export type BenchmarkMetric = "score" | "cost" | "duration";
export const BENCHMARK_METRICS: readonly BenchmarkMetric[] = ["score", "cost", "duration"];
/** Minimal evaluation shape needed to read a metric (BenchmarkEvaluation is a superset). */
/** Minimal Evaluation shape needed to read Score (BenchmarkEvaluation is a superset). */
export interface MetricSourceLike {
score: number;
cost?: number;
durationMs?: number;
}
/** Each evaluation's value under the selected metric; missing (cost / durationMs not recorded) is null (skipped point). */
export function metricValues(
evaluations: readonly MetricSourceLike[],
metric: BenchmarkMetric,
): (number | null)[] {
/** Each Evaluation's Score; non-finite malformed values become chart gaps defensively. */
export function scoreValues(evaluations: readonly MetricSourceLike[]): (number | null)[] {
return evaluations.map((e) => {
const v = metric === "score" ? e.score : metric === "cost" ? e.cost : e.durationMs;
return typeof v === "number" && Number.isFinite(v) ? v : null;
return typeof e.score === "number" && Number.isFinite(e.score) ? e.score : null;
});
}
@@ -63,37 +48,36 @@ export function metricMax(values: readonly (number | null)[]): number {
/** Minimal evaluation shape needed for series grouping (BenchmarkEvaluation is a superset). */
export interface ModelRefLike {
provider?: string;
modelId?: string;
thinkingLevel?: string;
}
/** One chart series: the set of evaluations sharing the same (provider, modelId) (older records with no model tag are grouped into a single series). */
/** One chart series: the set of evaluations sharing the same (modelId, thinkingLevel). */
export interface EvaluationSeries {
/** Grouping key (internal grouping only, not used as an id; empty string for untagged model). */
key: string;
provider?: string;
modelId?: string;
thinkingLevel?: string;
/** Matching evaluation indices: global time-axis positions, shared across all series on the same x-axis. */
indices: number[];
}
/**
* Groups evaluations into series by the model they carry: models are ordered by first
* appearance (color is picked from SERIES_COLORS by series index, so color follows the model and
* doesn't change with filtering); older records with no model tag are grouped into a trailing
* unnamed series (shown in gray, labeled "untagged model").
* Groups evaluations into series by model ID and thinking level, deliberately ignoring provider.
* Runtime combinations are ordered by first appearance (color is picked from SERIES_COLORS by
* series index); defensive records with no model tag are grouped into a trailing unnamed series.
*/
export function modelSeries(evaluations: readonly ModelRefLike[]): EvaluationSeries[] {
const map = new Map<string, EvaluationSeries>();
evaluations.forEach((e, index) => {
const labeled = e.modelId !== undefined && e.modelId !== "";
const key = labeled ? `${e.provider ?? ""}\u0000${e.modelId}` : "";
const key = labeled ? `${e.modelId}\u0000${e.thinkingLevel ?? ""}` : "";
let series = map.get(key);
if (!series) {
series = {
key,
...(labeled && e.provider !== undefined ? { provider: e.provider } : {}),
...(labeled && e.modelId !== undefined ? { modelId: e.modelId } : {}),
...(labeled && e.thinkingLevel !== undefined ? { thinkingLevel: e.thinkingLevel } : {}),
indices: [],
};
map.set(key, series);
@@ -104,12 +88,11 @@ export function modelSeries(evaluations: readonly ModelRefLike[]): EvaluationSer
return [...all.filter((x) => x.key !== ""), ...all.filter((x) => x.key === "")];
}
/** A series' value sequence under the selected metric: indices outside this series are null (skipped point), keeping the global time axis. */
/** A series' Score sequence; indices outside this series are null, keeping the global time axis. */
export function seriesValues(
evaluations: readonly (MetricSourceLike & ModelRefLike)[],
series: EvaluationSeries,
metric: BenchmarkMetric,
): (number | null)[] {
const own = new Set(series.indices);
return metricValues(evaluations, metric).map((v, i) => (own.has(i) ? v : null));
return scoreValues(evaluations).map((v, i) => (own.has(i) ? v : null));
}
@@ -1,21 +1,18 @@
/**
* Benchmark page (read-only display):
* the left directory lists Benchmarks grouped by Agent (the scoreboard is only fetched once
* expanded); the right side shows the selected Benchmark's title info, a chart (switches between
* score / cost / duration metrics on the same time axis; **grouped into series by the model each
* evaluation carries** — the model isn't part of benchmark_config, each evaluation carries its
* own — one color per series plus a legend, with older untagged records shown as a gray series;
* missing values are skipped points, breaking the line; reuses the usage center's ChartFrame
* coordinate system) and an evaluation detail table (includes a model column; rows expand to
* show the evaluation summary — title and body shown separately — and per-case scores, and case
* rows further expand to show the raw results of each run, with a Session link straight to that
* Session's trace observability).
* expanded); the right side shows the selected Benchmark's title info, a Score-only chart grouped
* into series by each Evaluation's model ID and thinking level, and an evaluation detail table
* with separate model ID and thinking-level columns. Rows expand to show the evaluation summary
* and per-case scores, and Case rows further expand to show the raw results of each Run with a
* Session link.
* With a ?agentId= deep link, only the target Agent is expanded by default.
*/
import { useEffect, useState } from "react";
import { Link, useSearchParams } from "react-router";
import type {
BenchmarkCaseScore,
BenchmarkCaseSummary,
BenchmarkEvaluation,
BenchmarkSummary,
} from "@prismshadow/penguin-server/api";
@@ -29,22 +26,22 @@ import { useTheme } from "../../state/theme";
import type { Currency } from "../../state/theme";
import { AgentAvatar } from "../../components/ui/agent-avatar";
import { Chevron } from "../../components/ui/chevron";
import { Segmented } from "../../components/ui/segmented";
import { Truncated } from "../../components/ui/truncated";
import { EmptyState } from "../../components/ui/empty-state";
import { Modal } from "../../components/ui/modal";
import { SkeletonList } from "../../components/ui/skeleton";
import { providerInfo } from "@prismshadow/penguin-core/model-catalog";
import { seriesColor } from "../../lib/category-colors";
import { makeGeom } from "../usage/chart-geom";
import { ChartFrame, useChartWidth } from "../usage/chart-svg";
import {
lineSegments,
metricMax,
metricValues,
modelSeries,
scoreValues,
seriesValues,
} from "./benchmark-metrics";
import type { BenchmarkMetric, EvaluationSeries } from "./benchmark-metrics";
import type { EvaluationSeries } from "./benchmark-metrics";
import { BenchmarkStatementBrowser } from "./benchmark-statement-browser";
interface Selection {
agentId: string;
@@ -144,61 +141,31 @@ function AgentNode({
);
}
/** Display label for a metric (shared by the Segmented options and the chart title; S is a live binding, must be read during render). */
function metricLabel(metric: BenchmarkMetric): string {
return metric === "score"
? S.benchmark.colScore
: metric === "cost"
? S.common.cost
: S.benchmark.colDuration;
}
/** Display format for a metric value (shared by y-axis ticks and the tooltip): score / cost / duration each have their own formatting rule. */
function formatMetric(metric: BenchmarkMetric, v: number, currency: Currency): string {
return metric === "score"
? formatScore(v)
: metric === "cost"
? formatMoney(v, currency)
: humanizeDuration(v);
}
/**
* Metric-over-time line chart (cloned from the usage center's TrendChart: area + line + data
* points, sharing the ChartFrame coordinate system). **Grouped into series** by the model each
* evaluation carries: one color per series (SERIES_COLORS is a fixed color sequence, color
* follows the model; older records with no model tag get a gray series), all series share the
* same time axis and y-axis range. Missing values within a series (indices outside this series,
* or cost / durationMs not recorded) are **skipped points** — lineSegments splits the
* value-bearing indices into segments, drawing area + line + data points within each segment and
* breaking between segments (a single-point segment draws only a point). ChartFrame's x-axis
* labels take slice(5) of dates: passing `yyyy-MM-dd HH:mm` displays as `MM-dd HH:mm`. A single
* evaluation is still drawn; no evaluations falls back to an empty state.
* Score-over-time line chart. Every Run, Case, and Evaluation uses the fixed 0..100 scale.
* Evaluations remain grouped by model ID and thinking level so a runtime change stays visible
* without adding other metric modes.
*/
function MetricTrendChart({
function ScoreTrendChart({
evaluations,
series,
metric,
currency,
}: {
evaluations: BenchmarkEvaluation[];
series: EvaluationSeries[];
metric: BenchmarkMetric;
currency: Currency;
}) {
const [hover, setHover] = useState<number | null>(null);
const [ref, width] = useChartWidth();
const values = metricValues(evaluations, metric);
const geom = makeGeom(evaluations.length, metricMax(values), width);
const values = scoreValues(evaluations);
const geom = makeGeom(evaluations.length, Math.max(metricMax(values), 100), width);
const dates = evaluations.map((e) => formatDateTime(e.time));
const baseY = geom.y(0);
return (
<div ref={ref}>
{width > 0 && (
<ChartFrame
geom={geom}
fmtY={(v) => formatMetric(metric, v, currency)}
fmtY={formatScore}
dates={dates}
hover={hover}
onHover={setHover}
@@ -209,18 +176,20 @@ function MetricTrendChart({
<>
<p className="text-gray-400">{formatDateTime(e.time)}</p>
<p className="font-mono">
{v === null ? "—" : formatMetric(metric, v, currency)}
{v === null ? "—" : formatScore(v)}
{e.version !== undefined && (
<span className="ml-1.5 text-gray-400">v{e.version}</span>
)}
</p>
{e.modelId && <p className="font-mono text-gray-400">{e.modelId}</p>}
<p className="font-mono text-gray-400">
{e.modelId} · {e.thinkingLevel}
</p>
</>
);
}}
>
{series.map((s, si) => {
const segments = lineSegments(seriesValues(evaluations, s, metric));
const segments = lineSegments(seriesValues(evaluations, s));
return (
<g
key={s.key === "" ? "unlabeled" : s.key}
@@ -230,18 +199,8 @@ function MetricTrendChart({
const line = seg
.map((p, j) => `${j === 0 ? "M" : "L"}${geom.x(p.index)},${geom.y(p.value)}`)
.join(" ");
const area = `${line} L${geom.x(seg[seg.length - 1]!.index)},${baseY} L${geom.x(seg[0]!.index)},${baseY} Z`;
return (
<g key={k}>
{/* Area fill: line closed to the baseline, low opacity reinforces the trend's sense of "volume" (no area for single-point segments) */}
{seg.length > 1 && (
<path
d={area}
className="fill-current"
stroke="none"
opacity={hover !== null ? 0.06 : 0.1}
/>
)}
{seg.length > 1 && (
<path
d={line}
@@ -274,55 +233,25 @@ function MetricTrendChart({
}
/**
* Chart section: title (follows metric) + metric switch (score / cost / duration, segmented
* control) + model legend (only shown with >=2 series; a single series' identity is carried by
* the detail table's model column and the hover tooltip instead) + the line chart. When the same
* modelId coexists across providers, the legend appends the provider's display name to
* disambiguate. Mounted under a keyed container per Benchmark: switching Benchmarks resets back
* to "score".
* Score chart + runtime legend. Provider is deliberately not part of chart identity.
*/
function TrendSection({
evaluations,
currency,
}: {
evaluations: BenchmarkEvaluation[];
currency: Currency;
}) {
const [metric, setMetric] = useState<BenchmarkMetric>("score");
function TrendSection({ evaluations }: { evaluations: BenchmarkEvaluation[] }) {
const series = modelSeries(evaluations);
const ids = series.map((s) => s.modelId).filter((v): v is string => v !== undefined);
const dupIds = new Set(ids.filter((id, i) => ids.indexOf(id) !== i));
const labelOf = (s: EvaluationSeries): string => {
if (!s.modelId) return S.benchmark.legendUnlabeled;
if (!dupIds.has(s.modelId)) return s.modelId;
const provider = s.provider ? (providerInfo(s.provider)?.label ?? s.provider) : "";
return provider ? `${s.modelId} · ${provider}` : s.modelId;
return s.thinkingLevel ? `${s.modelId} · ${s.thinkingLevel}` : s.modelId;
};
return (
<div>
<div className="mb-1 flex items-center justify-between gap-2">
<p className="text-xs font-semibold text-gray-500">
{S.benchmark.trendTitle(metricLabel(metric))}
</p>
<div className="w-44 shrink-0">
<Segmented
options={[
{ value: "score", label: metricLabel("score") },
{ value: "cost", label: metricLabel("cost") },
{ value: "duration", label: metricLabel("duration") },
]}
value={metric}
onChange={setMetric}
/>
</div>
</div>
<p className="mb-1 text-xs font-semibold text-gray-500">
{S.benchmark.trendTitle(S.benchmark.colScore)}
</p>
{series.length >= 2 && (
<div className="mb-1.5 flex flex-wrap items-center gap-x-3 gap-y-1">
{series.map((s, i) => (
<span
key={s.key === "" ? "unlabeled" : s.key}
className="flex items-center gap-1.5 text-[11px] text-gray-500 dark:text-gray-400"
title={s.provider ? (providerInfo(s.provider)?.label ?? s.provider) : undefined}
>
<span
className={`inline-block h-2 w-2 shrink-0 rounded-sm ${
@@ -334,26 +263,25 @@ function TrendSection({
))}
</div>
)}
<MetricTrendChart
evaluations={evaluations}
series={series}
metric={metric}
currency={currency}
/>
<ScoreTrendChart evaluations={evaluations} series={series} />
</div>
);
}
const CELL = "px-3 py-2";
/** One evaluation record: main row (time/version/total score/cost/duration) + a sub-table of per-case scores that expands on click. */
/** One evaluation record: main row + a sub-table of per-Case scores that expands on click. */
function EvaluationRow({
agentId,
evaluation,
caseTitles,
onOpenCase,
currency,
}: {
agentId: string;
evaluation: BenchmarkEvaluation;
caseTitles: ReadonlyMap<string, string>;
onOpenCase: (caseId: string) => void;
currency: Currency;
}) {
const [open, setOpen] = useState(false);
@@ -376,7 +304,10 @@ function EvaluationRow({
className={`${CELL} max-w-40 truncate font-mono text-xs text-gray-500 dark:text-gray-400`}
title={evaluation.provider}
>
{evaluation.modelId ?? "—"}
{evaluation.modelId}
</td>
<td className={`${CELL} font-mono text-xs text-gray-500 dark:text-gray-400`}>
{evaluation.thinkingLevel}
</td>
<td className={`${CELL} font-mono text-xs font-semibold tabular-nums`}>
{formatScore(evaluation.score)}
@@ -390,11 +321,8 @@ function EvaluationRow({
</tr>
{open && (
<tr className="border-b border-gray-100 last:border-b-0 dark:border-gray-800/60">
<td colSpan={6} className="bg-gray-50/80 px-3 py-2 dark:bg-gray-950/40">
{/* Evaluation summary (title + body shown separately; the generating side always
writes both, but the display side tolerates missing values — with an old-style
single-paragraph summary only, it's still shown as usual, prefixed with the
"Evaluation Summary" label). */}
<td colSpan={7} className="bg-gray-50/80 px-3 py-2 dark:bg-gray-950/40">
{/* Evaluation summary title and body are displayed separately when present. */}
{(evaluation.summaryTitle || evaluation.summary) && (
<div className="mb-2">
{evaluation.summaryTitle ? (
@@ -423,7 +351,14 @@ function EvaluationRow({
</thead>
<tbody>
{evaluation.cases.map((c) => (
<CaseRow key={c.case} agentId={agentId} caseScore={c} currency={currency} />
<CaseRow
key={c.case}
agentId={agentId}
caseScore={c}
title={caseTitles.get(c.case)}
onOpenCase={caseTitles.has(c.case) ? onOpenCase : undefined}
currency={currency}
/>
))}
</tbody>
</table>
@@ -449,33 +384,54 @@ function SessionLink({ agentId, sessionId }: { agentId: string; sessionId?: stri
}
/**
* Score row for one case: the case-level metrics = the average of its runs (already computed by
* the server, trust its values). With runs[] present, the row can expand to show the raw results
* of each run (#index + score / cost / duration / Session link); with the old format lacking
* runs, it's not expandable and the case-level single Session link is used as before.
* Score row for one Case: stored Case averages are authoritative. Expanding shows raw Run
* results; the UI never recomputes averages.
*/
function CaseRow({
agentId,
caseScore: c,
title,
onOpenCase,
currency,
}: {
agentId: string;
caseScore: BenchmarkCaseScore;
title?: string;
onOpenCase?: (caseId: string) => void;
currency: Currency;
}) {
const [open, setOpen] = useState(false);
const runs = c.runs ?? [];
const expandable = runs.length > 0;
const runs = c.runs;
return (
<>
<tr
onClick={expandable ? () => setOpen((v) => !v) : undefined}
className={`text-xs ${expandable ? "cursor-pointer transition-colors duration-150 hover:bg-gray-100/70 dark:hover:bg-gray-800/40" : ""}`}
onClick={() => setOpen((v) => !v)}
className="cursor-pointer text-xs transition-colors duration-150 hover:bg-gray-100/70 dark:hover:bg-gray-800/40"
>
<td className="px-2 py-1 font-mono">
<span className="flex items-center gap-1.5">
{expandable && <Chevron open={open} size={12} className="text-gray-400" />}
{c.case}
<td className="px-2 py-1">
<span className="flex items-start gap-1.5">
<Chevron open={open} size={12} className="text-gray-400" />
<span className="min-w-0">
{onOpenCase ? (
<button
type="button"
className="block text-left font-medium text-gray-800 hover:underline dark:text-gray-200"
onClick={(event) => {
event.stopPropagation();
onOpenCase(c.case);
}}
>
{title ?? c.case}
</button>
) : (
<span className="block font-medium text-gray-800 dark:text-gray-200">
{title ?? c.case}
</span>
)}
{title && title !== c.case && (
<span className="block font-mono text-[11px] text-gray-400">{c.case}</span>
)}
</span>
</span>
</td>
<td className="px-2 py-1 font-mono tabular-nums">{formatScore(c.score)}</td>
@@ -486,7 +442,7 @@ function CaseRow({
{c.durationMs !== undefined ? humanizeDuration(c.durationMs) : "—"}
</td>
<td className="px-2 py-1">
<SessionLink agentId={agentId} {...(c.sessionId ? { sessionId: c.sessionId } : {})} />
<span className="text-gray-400">—</span>
</td>
</tr>
{open &&
@@ -513,6 +469,51 @@ function CaseRow({
);
}
function CasesSection({
cases,
error,
onOpenCase,
}: {
cases: BenchmarkCaseSummary[] | null;
error: string | null;
onOpenCase: (caseId: string) => void;
}) {
return (
<div>
<p className="mb-1 text-xs font-semibold text-gray-500">{S.benchmark.cases}</p>
<div className="overflow-hidden rounded-md border border-gray-200 bg-white dark:border-gray-800 dark:bg-gray-900">
{error && <p className="px-3 py-2 text-xs text-red-500">{error}</p>}
{!cases && !error && <p className="px-3 py-2 text-xs text-gray-400">{S.common.loading}</p>}
{cases?.map((item) => {
return (
<button
key={item.id}
type="button"
onClick={() => onOpenCase(item.id)}
className="flex w-full items-center gap-3 border-b border-gray-100 px-3 py-2 text-left transition-colors last:border-b-0 hover:bg-gray-50 dark:border-gray-800/70 dark:hover:bg-gray-800/50"
>
<span className="min-w-0 flex-1">
<span className="block text-sm font-medium text-gray-800 dark:text-gray-200">
{item.title}
</span>
<span className="block truncate font-mono text-[11px] text-gray-400">
{item.id}
</span>
</span>
<span className="shrink-0 font-mono text-xs text-gray-500">
{S.benchmark.maxScore("100")}
</span>
<span className="shrink-0 text-xs text-brand-700 dark:text-brand-300">
{S.benchmark.viewCase}
</span>
</button>
);
})}
</div>
</div>
);
}
export function BenchmarkPage() {
useDocumentTitle(S.benchmark.title);
const { currentProject, agents, agentsLoading } = useProject();
@@ -522,17 +523,42 @@ export function BenchmarkPage() {
const [searchParams] = useSearchParams();
const focusAgentId = searchParams.get("agentId");
const [selection, setSelection] = useState<Selection | null>(null);
const [caseStatements, setCaseStatements] = useState<BenchmarkCaseSummary[] | null>(null);
const [caseError, setCaseError] = useState<string | null>(null);
const [openCaseId, setOpenCaseId] = useState<string | null>(null);
// Clear the selection when the Project changes.
useEffect(() => {
setSelection(null);
}, [projectId]);
useEffect(() => {
setCaseStatements(null);
setCaseError(null);
setOpenCaseId(null);
if (!projectId || !selection) return;
let cancelled = false;
api
.listBenchmarkCases(projectId, selection.agentId, selection.benchmark.id)
.then((data) => {
if (!cancelled) setCaseStatements(data.cases);
})
.catch((error: unknown) => {
if (!cancelled) setCaseError(apiErrorText(error));
});
return () => {
cancelled = true;
};
}, [projectId, selection]);
if (!projectId) return null;
const bm = selection?.benchmark ?? null;
// Chart uses ascending time order (the scoreboard is already ordered, this sort is defensive); the detail table shows newest first.
const evaluations = bm ? [...bm.evaluations].sort((a, b) => a.time.localeCompare(b.time)) : [];
// The Scoreboard append order is the evaluation sequence. Preserve it even when a malformed
// timestamp would otherwise reorder Agent versions; the detail table shows that sequence newest first.
const evaluations = bm ? [...bm.evaluations] : [];
const caseTitles = new Map(caseStatements?.map((item) => [item.id, item.title]) ?? []);
const openCase = caseStatements?.find((item) => item.id === openCaseId) ?? null;
return (
<div className="flex h-full flex-col md:flex-row">
@@ -564,9 +590,7 @@ export function BenchmarkPage() {
{selection && bm ? (
// Changing the key on Benchmark switch resets expand state (a detail row's open doesn't linger across Benchmarks).
<div key={`${selection.agentId}/${bm.id}`} className="mx-auto max-w-4xl space-y-4">
{/* Title row: title + case count (the model isn't part of config — each evaluation
carries its own, see the chart legend and the detail table's model column) +
description */}
{/* Runtime belongs to each Evaluation and is shown in the detail table. */}
<div>
<div className="flex flex-wrap items-baseline gap-x-2 gap-y-1">
<h1 className="min-w-0 truncate text-lg font-semibold">{bm.title}</h1>
@@ -577,23 +601,26 @@ export function BenchmarkPage() {
)}
</div>
<CasesSection cases={caseStatements} error={caseError} onOpenCase={setOpenCaseId} />
{evaluations.length === 0 ? (
<EmptyState title={S.benchmark.noEvaluations} />
) : (
<>
<TrendSection evaluations={evaluations} currency={currency} />
<TrendSection evaluations={evaluations} />
<div>
<p className="mb-1 text-xs font-semibold text-gray-500">
{S.benchmark.evaluations}
</p>
<div className="overflow-x-auto overflow-y-clip rounded-md border border-gray-200 bg-white dark:border-gray-800 dark:bg-gray-900">
<table className="w-full min-w-[600px] text-left text-sm">
<table className="w-full min-w-[720px] text-left text-sm">
<thead>
<tr className="border-b border-gray-200 bg-gray-50/80 text-xs text-gray-500 dark:border-gray-800 dark:bg-gray-900">
<th className="px-3 py-2.5">{S.common.time}</th>
<th className="px-3 py-2.5">{S.benchmark.colVersion}</th>
<th className="px-3 py-2.5">{S.benchmark.colModel}</th>
<th className="px-3 py-2.5">{S.benchmark.colThinkingLevel}</th>
<th className="px-3 py-2.5">{S.benchmark.colScore}</th>
<th className="px-3 py-2.5">{S.common.cost}</th>
<th className="px-3 py-2.5">{S.benchmark.colDuration}</th>
@@ -605,6 +632,8 @@ export function BenchmarkPage() {
key={i}
agentId={selection.agentId}
evaluation={ev}
caseTitles={caseTitles}
onOpenCase={setOpenCaseId}
currency={currency}
/>
))}
@@ -614,6 +643,21 @@ export function BenchmarkPage() {
</div>
</>
)}
{openCase && (
<Modal
open
title={openCase.title}
widthClass="sm:max-w-6xl"
onClose={() => setOpenCaseId(null)}
>
<BenchmarkStatementBrowser
projectId={projectId}
agentId={selection.agentId}
benchmarkId={bm.id}
caseSummary={openCase}
/>
</Modal>
)}
</div>
) : (
<EmptyState title={S.benchmark.selectBenchmark} />
@@ -0,0 +1,367 @@
import { useCallback, useEffect, useRef, useState } from "react";
import type {
BenchmarkCaseSummary,
WorkspaceFileEntry,
WorkspaceFilesResponse,
} from "@prismshadow/penguin-server/api";
import ReactMarkdown from "react-markdown";
import type { Components } from "react-markdown";
import remarkGfm from "remark-gfm";
import * as api from "../../api/endpoints";
import { apiErrorText } from "../../lib/api-error";
import { formatBytes } from "../../lib/format";
import { S } from "../../lib/strings";
import { SkeletonList } from "../../components/ui/skeleton";
import { CodeBlock } from "../chat/code-block";
const TEXT_EXTS = new Set([
"txt",
"md",
"json",
"js",
"mjs",
"cjs",
"ts",
"tsx",
"jsx",
"py",
"sh",
"bash",
"yaml",
"yml",
"toml",
"css",
"html",
"htm",
"csv",
"log",
"xml",
"ini",
"conf",
"sql",
"svg",
]);
const IMAGE_EXTS = new Set(["png", "jpg", "jpeg", "gif", "webp"]);
const EXTERNAL_REF_RE = /^[a-z][a-z0-9+.-]*:/i;
const HIGHLIGHT_LIMIT = 64 * 1024;
interface Preview {
path: string;
name: string;
kind: "text" | "md" | "image" | "pdf" | "unsupported";
content?: string;
truncated?: boolean;
loading?: boolean;
error?: string;
}
interface Props {
projectId: string;
agentId: string;
benchmarkId: string;
caseSummary: BenchmarkCaseSummary;
}
function extOf(name: string): string {
const index = name.lastIndexOf(".");
return index >= 0 ? name.slice(index + 1).toLowerCase() : name.toLowerCase();
}
function joinPath(dir: string, name: string): string {
return dir === "" ? name : `${dir}/${name}`;
}
function dirOf(filePath: string): string {
return filePath.includes("/") ? filePath.slice(0, filePath.lastIndexOf("/")) : "";
}
function resolveRelative(baseDir: string, ref: string): string {
const out = ref.startsWith("/") || baseDir === "" ? [] : baseDir.split("/");
for (const segment of ref.split("/")) {
if (segment === "" || segment === ".") continue;
if (segment === "..") out.pop();
else out.push(segment);
}
return out.join("/");
}
function languageFor(name: string): string {
const ext = extOf(name);
return (
{
json: "json",
md: "markdown",
js: "javascript",
mjs: "javascript",
cjs: "javascript",
jsx: "jsx",
ts: "typescript",
tsx: "tsx",
py: "python",
sh: "shellscript",
bash: "shellscript",
yaml: "yaml",
yml: "yaml",
toml: "toml",
css: "css",
html: "html",
htm: "html",
xml: "xml",
sql: "sql",
svg: "xml",
}[ext] ?? "text"
);
}
export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, caseSummary }: Props) {
const [path, setPath] = useState("");
const [listing, setListing] = useState<WorkspaceFilesResponse | null>(null);
const [listError, setListError] = useState<string | null>(null);
const [preview, setPreview] = useState<Preview | null>(null);
const initialReadmeOpened = useRef(false);
const previewRequest = useRef(0);
const fileUrl = useCallback(
(filePath: string, options?: { download?: boolean; preview?: boolean }) =>
api.benchmarkCaseFileUrl(projectId, agentId, benchmarkId, caseSummary.id, filePath, options),
[projectId, agentId, benchmarkId, caseSummary.id],
);
const previewPath = useCallback(
async (filePath: string) => {
const request = ++previewRequest.current;
const name = filePath.includes("/")
? filePath.slice(filePath.lastIndexOf("/") + 1)
: filePath;
const ext = extOf(name);
if (IMAGE_EXTS.has(ext)) {
setPreview({ path: filePath, name, kind: "image" });
return;
}
if (ext === "pdf") {
setPreview({ path: filePath, name, kind: "pdf" });
return;
}
const isMarkdown = ext === "md";
if (!TEXT_EXTS.has(ext)) {
setPreview({ path: filePath, name, kind: "unsupported" });
return;
}
setPreview({ path: filePath, name, kind: isMarkdown ? "md" : "text", loading: true });
try {
const response = await fetch(fileUrl(filePath, { preview: true }), {
credentials: "same-origin",
});
if (!response.ok) throw new Error(String(response.status));
const content = await response.text();
if (request !== previewRequest.current) return;
setPreview({
path: filePath,
name,
kind: isMarkdown ? "md" : "text",
content,
truncated: response.headers.get("x-content-truncated") === "1",
});
} catch (error) {
if (request !== previewRequest.current) return;
setPreview({
path: filePath,
name,
kind: isMarkdown ? "md" : "text",
error: apiErrorText(error),
});
}
},
[fileUrl],
);
useEffect(() => {
setListing(null);
setListError(null);
let cancelled = false;
api
.listBenchmarkCaseFiles(projectId, agentId, benchmarkId, caseSummary.id, path)
.then((data) => {
if (cancelled) return;
setListing(data);
if (path === "" && !initialReadmeOpened.current) {
initialReadmeOpened.current = true;
const readme = data.entries.find(
(entry) => entry.kind === "file" && entry.name.toLowerCase() === "readme.md",
);
if (readme) void previewPath(readme.name);
}
})
.catch((error: unknown) => {
if (!cancelled) setListError(apiErrorText(error));
});
return () => {
cancelled = true;
};
}, [projectId, agentId, benchmarkId, caseSummary.id, path, previewPath]);
const crumbs = path === "" ? [] : path.split("/");
const downloadUrl = preview ? fileUrl(preview.path, { download: true }) : null;
const openEntry = (entry: WorkspaceFileEntry) => {
if (entry.kind === "dir") {
setPath(joinPath(path, entry.name));
return;
}
void previewPath(joinPath(path, entry.name));
};
const markdownComponents: Components = {
img: ({ src, alt }) => (
<img
src={
typeof src === "string" && !EXTERNAL_REF_RE.test(src)
? fileUrl(resolveRelative(dirOf(preview?.path ?? ""), src))
: src
}
alt={alt ?? ""}
loading="lazy"
className="max-w-full"
/>
),
a: ({ href, children }) => {
if (typeof href !== "string" || href.startsWith("#")) return <a href={href}>{children}</a>;
if (EXTERNAL_REF_RE.test(href)) {
return (
<a href={href} target="_blank" rel="noreferrer">
{children}
</a>
);
}
const target = resolveRelative(dirOf(preview?.path ?? ""), href);
return (
<a
href={fileUrl(target)}
onClick={(event) => {
event.preventDefault();
void previewPath(target);
}}
>
{children}
</a>
);
},
};
return (
<div className="grid min-h-[58vh] overflow-hidden rounded-md border border-gray-200 md:grid-cols-[240px_minmax(0,1fr)] dark:border-gray-800">
<aside className="border-b border-gray-200 bg-gray-50/60 md:border-b-0 md:border-r dark:border-gray-800 dark:bg-gray-950/30">
<div className="flex flex-wrap items-center gap-1 border-b border-gray-200 px-2 py-2 dark:border-gray-800">
<button
type="button"
onClick={() => setPath("")}
className="rounded px-1.5 py-0.5 text-xs text-gray-600 hover:bg-gray-100 dark:text-gray-300 dark:hover:bg-gray-800"
>
{S.benchmark.publicMaterials}
</button>
{crumbs.map((segment, index) => (
<span key={`${segment}-${index}`} className="flex min-w-0 items-center gap-1">
<span className="text-gray-300 dark:text-gray-700">/</span>
<button
type="button"
onClick={() => setPath(crumbs.slice(0, index + 1).join("/"))}
className="max-w-24 truncate rounded px-1 py-0.5 text-xs text-gray-600 hover:bg-gray-100 dark:text-gray-300 dark:hover:bg-gray-800"
>
{segment}
</button>
</span>
))}
</div>
<div className="max-h-44 overflow-y-auto md:max-h-[53vh]">
{listError && <p className="px-3 py-2 text-xs text-red-500">{listError}</p>}
{!listing && !listError && <SkeletonList rows={4} />}
{listing?.entries.length === 0 && (
<p className="px-3 py-2 text-xs text-gray-400">{S.files.empty}</p>
)}
{listing?.entries.map((entry) => (
<button
key={`${entry.kind}/${entry.name}`}
type="button"
onClick={() => openEntry(entry)}
className="flex w-full items-center gap-2 border-b border-gray-100 px-3 py-2 text-left hover:bg-gray-100 dark:border-gray-800/70 dark:hover:bg-gray-800/60"
>
<span className="text-sm text-gray-400">{entry.kind === "dir" ? "▸" : "·"}</span>
<span className="min-w-0 flex-1 truncate text-sm">{entry.name}</span>
{entry.kind === "file" && (
<span className="shrink-0 text-[11px] text-gray-400">
{formatBytes(entry.sizeBytes)}
</span>
)}
</button>
))}
</div>
</aside>
<section className="min-w-0">
<div className="flex min-h-11 flex-wrap items-center gap-2 border-b border-gray-200 px-3 py-2 dark:border-gray-800">
<div className="min-w-0 flex-1">
<p className="truncate font-mono text-xs text-gray-500">
{preview?.path ?? caseSummary.id}
</p>
</div>
<span className="shrink-0 text-xs text-gray-400">{S.benchmark.maxScore("100")}</span>
{downloadUrl && preview && (
<a
href={downloadUrl}
download={preview.name}
className="rounded-md px-2.5 py-1 text-xs font-medium text-gray-600 hover:bg-gray-100 dark:text-gray-300 dark:hover:bg-gray-800"
>
{S.files.download}
</a>
)}
</div>
<div className="max-h-[52vh] min-h-[52vh] overflow-auto p-3">
{!preview ? (
<p className="text-sm text-gray-400">{S.benchmark.statementUnavailable}</p>
) : preview.loading ? (
<SkeletonList rows={8} />
) : preview.error ? (
<p className="text-sm text-red-500">{preview.error}</p>
) : preview.kind === "image" ? (
<img
src={fileUrl(preview.path)}
alt={preview.name}
loading="lazy"
className="max-w-full rounded-md border border-gray-200 dark:border-gray-800"
/>
) : preview.kind === "pdf" ? (
<iframe
src={fileUrl(preview.path)}
title={preview.name}
className="h-[50vh] w-full rounded-md border border-gray-200 dark:border-gray-800"
/>
) : preview.kind === "md" ? (
<>
<div className="md-body text-sm text-gray-800 dark:text-gray-100">
<ReactMarkdown remarkPlugins={[remarkGfm]} components={markdownComponents}>
{preview.content ?? ""}
</ReactMarkdown>
</div>
{preview.truncated && (
<p className="mt-2 text-xs text-gray-400">{S.files.previewTruncated}</p>
)}
</>
) : preview.kind === "text" ? (
<>
<CodeBlock
language={languageFor(preview.name)}
code={preview.content ?? ""}
highlight={(preview.content?.length ?? 0) <= HIGHLIGHT_LIMIT}
/>
{preview.truncated && (
<p className="mt-2 text-xs text-gray-400">{S.files.previewTruncated}</p>
)}
</>
) : (
<p className="text-sm text-gray-500 dark:text-gray-400">{S.files.previewUnsupported}</p>
)}
</div>
</section>
</div>
);
}
+22 -15
View File
@@ -55,6 +55,7 @@ import { toastError } from "../../components/ui/toast";
import { useVersionInfo } from "../../lib/use-version-info";
import { ChatInput } from "./chat-input";
import { buildSkillsMessage } from "./skill-use";
import { EXAMPLE_TASKS, type ExampleTask, type ExampleTaskId } from "./example-tasks";
import { clearDraft, draftKey, loadDraft, saveDraft } from "./draft-cache";
import type { DraftCache } from "./draft-cache";
import { sameModelRef } from "../models/model-grouping";
@@ -88,18 +89,6 @@ function saveAppliedRouteKey(field: RouteStateField, key: string): void {
}
}
/**
* Example tasks on the draft screen, in display order (game card first, LoL music player,
* then the RAG build). Copy lives in S.chat.exampleTasks[id]; skills are pinned via a
* `[use_skills]` block — only those the selected Agent actually has installed are included,
* so the block never references a skill the agent can't read.
*/
const EXAMPLE_TASKS: { id: "game" | "lol" | "rag"; skills: string[] }[] = [
{ id: "game", skills: ["web-design"] },
{ id: "lol", skills: ["web-design"] },
{ id: "rag", skills: ["penguin-sdk", "web-design"] },
];
export function DraftView({
projectId,
models,
@@ -415,9 +404,9 @@ export function DraftView({
// busy id drives the clicked card's spinner; the shared in-flight guard and all failure
// handling live in onSend). keepDraft: an example never consumes the composer text, so a
// typed-but-unsent draft must survive. The selected model / Workspace / approval mode apply as-is.
const [exampleBusy, setExampleBusy] = useState<"game" | "lol" | "rag" | null>(null);
const [exampleBusy, setExampleBusy] = useState<ExampleTaskId | null>(null);
const runExample = useCallback(
async (task: (typeof EXAMPLE_TASKS)[number]) => {
async (task: ExampleTask) => {
if (exampleBusy !== null) return;
setExampleBusy(task.id);
try {
@@ -517,7 +506,25 @@ export function DraftView({
className="group flex min-w-0 items-center gap-3 rounded-xl border border-gray-200 bg-white px-4 py-3 text-left transition-colors duration-150 hover:border-gray-300 disabled:cursor-default disabled:opacity-60 dark:border-gray-800 dark:bg-gray-900 dark:hover:border-gray-700"
>
{/* 24×24 line icons (gamepad / music note / sparkle), consistent with the icon convention */}
{task.id === "lol" ? (
{task.id === "agentBenchmarkBuild" || task.id === "agentOptimization" ? (
<svg
width="20"
height="20"
viewBox="0 0 24 24"
fill="none"
stroke="currentColor"
className="shrink-0 text-brand-500 dark:text-brand-400"
aria-hidden
>
<path
d="M7 7h8a4 4 0 0 1 4 4v1M17 5l2 2-2 2M17 17H9a4 4 0 0 1-4-4v-1M7 19l-2-2 2-2"
strokeWidth="1.7"
strokeLinecap="round"
strokeLinejoin="round"
/>
<circle cx="12" cy="12" r="2" strokeWidth="1.7" />
</svg>
) : task.id === "lol" ? (
<svg
width="20"
height="20"
@@ -0,0 +1,17 @@
/**
* Draft-screen example cards in display order.
*
* Copy and full prompts live in the active locale dictionary at
* `S.chat.exampleTasks[id]`. Skills listed here are pinned only when the
* selected Agent has them installed; an empty list sends the prompt unchanged.
*/
export const EXAMPLE_TASKS = [
{ id: "game", skills: ["web-design"] },
{ id: "lol", skills: ["web-design"] },
{ id: "rag", skills: ["penguin-sdk", "web-design"] },
{ id: "agentBenchmarkBuild", skills: [] },
{ id: "agentOptimization", skills: [] },
] as const;
export type ExampleTask = (typeof EXAMPLE_TASKS)[number];
export type ExampleTaskId = ExampleTask["id"];
+2 -2
View File
@@ -39,8 +39,8 @@ export interface SeriesColor {
}
/**
* Fixed color sequence for line series (the eval center splits series by
* model): maintained separately from CATEGORY_COLORS — adjacent line colors
* Fixed color sequence for line series (the eval center splits series by model ID and thinking
* level): maintained separately from CATEGORY_COLORS — adjacent line colors
* must pass color-vision-deficiency (CVD) separation checks and dark shades
* must land within the brightness band. The violet / amber / sky / rose
* ordering passes the dataviz validator in both modes (dark-mode amber / sky
+2 -2
View File
@@ -126,9 +126,9 @@ export function formatMoney(
return `${symbol}${v.toFixed(digits)}`;
}
/** Benchmark total score display: integers unchanged, decimals keep one place (full-score convention is defined per-Benchmark by its scoring rubric). */
/** Benchmark score display: integers unchanged, decimals keep up to the stored two places. */
export function formatScore(n: number): string {
return Number.isInteger(n) ? `${n}` : trimZero(n);
return Number.isInteger(n) ? `${n}` : n.toFixed(2).replace(/0+$/, "").replace(/\.$/, "");
}
/** True while the one-decimal rendering of `v` still fits its unit: 1023.9 does, 1023.99 does not. */
+40 -1
View File
@@ -604,6 +604,39 @@ When done, open index.html in a browser and self-test once.`,
"give it a beautiful web chat UI following the web-design skill, with a few example questions in the empty state. " +
"When done, run the app, verify one streamed answer yourself, and tell me how to access it.",
},
agentBenchmarkBuild: {
label: "Example: create a decision Agent and capability evaluation",
desc: "Create a general decision Agent and test it on football, after-sales, and investment tasks",
prompt: `Use \`agent-creation\` followed by \`benchmark-design\` to create a decision Agent and produce a frozen Benchmark with a Formal Baseline.
Agent:
- id: \`finite_choice_agent\`
- capability: make stable, explainable finite choices when public information is incomplete or conflicting
- installed_skills: \`[]\`
Benchmark:
- id: \`contextual-choice-adaptation\`
- capability: form and transfer a stable finite-choice decision process from public rules, historical examples, and current facts
- runs: \`1\`
- desired_baseline_score: \`<75\`
- pilot_iteration_limit: \`5\`
Scenarios:
1. Make football betting decisions from historical matches and current information.
2. Choose after-sales actions from policy and ticket facts.
3. Choose investment actions from a strategy, historical markets, and current indicators.`,
},
agentOptimization: {
label: "Example: optimize a decision Agent from its evaluation",
desc: "Improve an Agent from existing evaluation results and verify that the new version is better",
prompt: `Use \`agent-optimization\` to optimize a decision Agent against its frozen Benchmark.
- test_agent_id: \`finite_choice_agent\`
- benchmark_id: \`contextual-choice-adaptation\`
- capability_direction: improve stability under incomplete information, conflicting rules, and finite choices
- desired_score: \`>=95\`
- candidate_round_limit: \`5\``,
},
},
sessionList: "Sessions",
defaultSessionTitle: "New chat",
@@ -944,12 +977,18 @@ When done, open index.html in a browser and self-test once.`,
emptyAgent: "No Benchmarks for this agent",
caseCount: (n: number): string => `${n} case${n === 1 ? "" : "s"}`,
trendTitle: (metric: string): string => `${metric} over time`,
cases: "Cases",
viewCase: "View task",
publicMaterials: "Public materials",
statementUnavailable: "The public Statement is unavailable",
maxScore: (score: string): string => `Max ${score}`,
evaluations: "Evaluations",
noEvaluations: "No evaluations yet",
summaryLabel: "Summary",
legendUnlabeled: "unlabeled model",
colVersion: "Version",
colModel: "Model",
colModel: "Model ID",
colThinkingLevel: "Thinking level",
colScore: "Score",
colDuration: "Duration",
colCase: "Case",
+45 -8
View File
@@ -532,11 +532,9 @@ export const zh = {
newSessionInWorkspace: "在此工作区新建对话",
draftSubtitle: "最擅长 AI 开发任务的自进化 Agent",
/**
* Example task cards on the draft screen: one click auto-submits the canned prompt (game
* card, then the LoL-player card, then the RAG card). These are the FULL working prompts — the README and
* landing page show a condensed one-sentence version of the RAG example for reading, and
* the cards' own desc lines stay short, but what actually gets submitted stays detailed:
* build quality depends on it.
* Example task cards on the draft screen: one click auto-submits the canned prompt. These
* are the FULL working prompts — descriptions stay short, but the submitted instructions
* remain detailed because execution quality depends on them.
*/
exampleTasks: {
game: {
@@ -590,6 +588,39 @@ Penguin 视觉风格(见 web-design 技能),深色/浅色主题(<html da
"按 web-design 技能提供美观的 Web 聊天界面,空态展示几个示例问题。" +
"完成后运行应用、自测一个问题验证流式回答,并告诉我访问方式。",
},
agentBenchmarkBuild: {
label: "示例:创建决策 Agent 和能力评测",
desc: "创建一个通用决策 Agent,并用足球、售后和投资任务检验它",
prompt: `请依次使用 \`agent-creation\` 和 \`benchmark-design\`,创建决策 Agent,并产出 Frozen Benchmark 与 Formal Baseline。
Agent:
- id:\`finite_choice_agent\`
- 能力:面对有限选项,在公开信息不足或冲突时仍能给出稳定、可解释的选择
- installed_skills:\`[]\`
Benchmark:
- id:\`contextual-choice-adaptation\`
- capability:从公开规则、历史案例和当前事实中形成并迁移稳定的有限选择决策过程
- runs:\`1\`
- desired_baseline_score:\`<75\`
- pilot_iteration_limit:\`5\`
场景:
1. 根据历史比赛与当前信息进行足球投注决策。
2. 根据售后政策与工单事实选择处置动作。
3. 根据投资策略、历史市场与当前指标选择投资动作。`,
},
agentOptimization: {
label: "示例:根据评测优化决策 Agent",
desc: "根据已有评测结果改进 Agent,并验证新版本是否真正提升",
prompt: `请使用 \`agent-optimization\`,根据 Frozen Benchmark 优化决策 Agent。
- test_agent_id:\`finite_choice_agent\`
- benchmark_id:\`contextual-choice-adaptation\`
- capability_direction:提高信息不完整、规则冲突和有限选项决策中的稳定性
- desired_score:\`>=95\`
- candidate_round_limit:\`5\``,
},
},
sessionList: "Session",
defaultSessionTitle: "新对话",
@@ -923,8 +954,13 @@ Penguin 视觉风格(见 web-design 技能),深色/浅色主题(<html da
selectBenchmark: "在左侧选择一个 Benchmark",
emptyAgent: "该 Agent 暂无 Benchmark",
caseCount: (n: number): string => `${n} 题`,
/** Chart title, varies by selected metric (score / cost / duration over time). */
/** Score-only chart title. */
trendTitle: (metric: string): string => `${metric}随时间变化`,
cases: "题目",
viewCase: "查看题目",
publicMaterials: "公开材料",
statementUnavailable: "题面暂时无法读取",
maxScore: (score: string): string => `满分 ${score}`,
evaluations: "评估明细",
noEvaluations: "暂无评估记录",
/** Evaluation notes (scoreboard's summary: score source and notes on this round's changes). */
@@ -932,8 +968,9 @@ Penguin 视觉风格(见 web-design 技能),深色/浅色主题(<html da
/** Chart legend: older evaluation records with no model label (gray series). */
legendUnlabeled: "未标注模型",
colVersion: "版本",
colModel: "模型",
colScore: "总分",
colModel: "模型 ID",
colThinkingLevel: "推理强度",
colScore: "Score",
colDuration: "耗时",
colCase: "题目",
colRun: "运行",