feat: 完善两阶段 Agent 自进化 Pipeline、Skills 与 Benchmark (#129)
This commit is contained in:
@@ -21,6 +21,7 @@ import type {
|
||||
ApprovalDecisionRequest,
|
||||
AuthLoginRequest,
|
||||
AuthResponse,
|
||||
BenchmarkCasesResponse,
|
||||
BenchmarksResponse,
|
||||
DirListResponse,
|
||||
FilesStatRequest,
|
||||
@@ -522,6 +523,44 @@ export const listBenchmarks = (projectId: string, agentId: string) =>
|
||||
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}/benchmarks`,
|
||||
);
|
||||
|
||||
export const listBenchmarkCases = (projectId: string, agentId: string, benchmarkId: string) =>
|
||||
apiFetch<BenchmarkCasesResponse>(
|
||||
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` +
|
||||
`/benchmarks/${encodeURIComponent(benchmarkId)}/cases`,
|
||||
);
|
||||
|
||||
export const listBenchmarkCaseFiles = (
|
||||
projectId: string,
|
||||
agentId: string,
|
||||
benchmarkId: string,
|
||||
caseId: string,
|
||||
path: string,
|
||||
) =>
|
||||
apiFetch<WorkspaceFilesResponse>(
|
||||
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` +
|
||||
`/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}/files`,
|
||||
{ query: { path } },
|
||||
);
|
||||
|
||||
export const benchmarkCaseFileUrl = (
|
||||
projectId: string,
|
||||
agentId: string,
|
||||
benchmarkId: string,
|
||||
caseId: string,
|
||||
path: string,
|
||||
options?: { download?: boolean; preview?: boolean },
|
||||
): string => {
|
||||
const base =
|
||||
`/api/projects/${encodeURIComponent(projectId)}/agents/${encodeURIComponent(agentId)}` +
|
||||
`/benchmarks/${encodeURIComponent(benchmarkId)}/cases/${encodeURIComponent(caseId)}` +
|
||||
`/files/content?path=${encodeURIComponent(path)}`;
|
||||
return (
|
||||
base +
|
||||
(options?.download ? "&download=1" : "") +
|
||||
(options?.preview && !options.download ? "&preview=1" : "")
|
||||
);
|
||||
};
|
||||
|
||||
// Agent State snapshot export / import ------------------------------------------------------
|
||||
|
||||
/** Snapshot bundle (tar.gz) download URL: the server sets Content-Disposition attachment, usable directly in <a download>. */
|
||||
|
||||
@@ -1,32 +1,17 @@
|
||||
/**
|
||||
* Metric switching and per-model series grouping for the Benchmark center chart (pure functions,
|
||||
* easy to unit test): the chart can switch between the score / cost / duration metrics (sharing
|
||||
* the same time axis), and is grouped into series by each evaluation's (provider, modelId) — one
|
||||
* color per series, legend by model. score is always present;
|
||||
* cost / durationMs are optional — missing values are **skipped points**: neither drawn nor
|
||||
* connected, so the line breaks at the gap (lineSegments splits value-bearing indices into
|
||||
* contiguous segments).
|
||||
* Score-only data helpers and per-runtime series grouping for the Benchmark center chart.
|
||||
* Series share one time axis and use each Evaluation's authoritative stored Score.
|
||||
*/
|
||||
|
||||
export type BenchmarkMetric = "score" | "cost" | "duration";
|
||||
|
||||
export const BENCHMARK_METRICS: readonly BenchmarkMetric[] = ["score", "cost", "duration"];
|
||||
|
||||
/** Minimal evaluation shape needed to read a metric (BenchmarkEvaluation is a superset). */
|
||||
/** Minimal Evaluation shape needed to read Score (BenchmarkEvaluation is a superset). */
|
||||
export interface MetricSourceLike {
|
||||
score: number;
|
||||
cost?: number;
|
||||
durationMs?: number;
|
||||
}
|
||||
|
||||
/** Each evaluation's value under the selected metric; missing (cost / durationMs not recorded) is null (skipped point). */
|
||||
export function metricValues(
|
||||
evaluations: readonly MetricSourceLike[],
|
||||
metric: BenchmarkMetric,
|
||||
): (number | null)[] {
|
||||
/** Each Evaluation's Score; non-finite malformed values become chart gaps defensively. */
|
||||
export function scoreValues(evaluations: readonly MetricSourceLike[]): (number | null)[] {
|
||||
return evaluations.map((e) => {
|
||||
const v = metric === "score" ? e.score : metric === "cost" ? e.cost : e.durationMs;
|
||||
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
||||
return typeof e.score === "number" && Number.isFinite(e.score) ? e.score : null;
|
||||
});
|
||||
}
|
||||
|
||||
@@ -63,37 +48,36 @@ export function metricMax(values: readonly (number | null)[]): number {
|
||||
|
||||
/** Minimal evaluation shape needed for series grouping (BenchmarkEvaluation is a superset). */
|
||||
export interface ModelRefLike {
|
||||
provider?: string;
|
||||
modelId?: string;
|
||||
thinkingLevel?: string;
|
||||
}
|
||||
|
||||
/** One chart series: the set of evaluations sharing the same (provider, modelId) (older records with no model tag are grouped into a single series). */
|
||||
/** One chart series: the set of evaluations sharing the same (modelId, thinkingLevel). */
|
||||
export interface EvaluationSeries {
|
||||
/** Grouping key (internal grouping only, not used as an id; empty string for untagged model). */
|
||||
key: string;
|
||||
provider?: string;
|
||||
modelId?: string;
|
||||
thinkingLevel?: string;
|
||||
/** Matching evaluation indices: global time-axis positions, shared across all series on the same x-axis. */
|
||||
indices: number[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Groups evaluations into series by the model they carry: models are ordered by first
|
||||
* appearance (color is picked from SERIES_COLORS by series index, so color follows the model and
|
||||
* doesn't change with filtering); older records with no model tag are grouped into a trailing
|
||||
* unnamed series (shown in gray, labeled "untagged model").
|
||||
* Groups evaluations into series by model ID and thinking level, deliberately ignoring provider.
|
||||
* Runtime combinations are ordered by first appearance (color is picked from SERIES_COLORS by
|
||||
* series index); defensive records with no model tag are grouped into a trailing unnamed series.
|
||||
*/
|
||||
export function modelSeries(evaluations: readonly ModelRefLike[]): EvaluationSeries[] {
|
||||
const map = new Map<string, EvaluationSeries>();
|
||||
evaluations.forEach((e, index) => {
|
||||
const labeled = e.modelId !== undefined && e.modelId !== "";
|
||||
const key = labeled ? `${e.provider ?? ""}\u0000${e.modelId}` : "";
|
||||
const key = labeled ? `${e.modelId}\u0000${e.thinkingLevel ?? ""}` : "";
|
||||
let series = map.get(key);
|
||||
if (!series) {
|
||||
series = {
|
||||
key,
|
||||
...(labeled && e.provider !== undefined ? { provider: e.provider } : {}),
|
||||
...(labeled && e.modelId !== undefined ? { modelId: e.modelId } : {}),
|
||||
...(labeled && e.thinkingLevel !== undefined ? { thinkingLevel: e.thinkingLevel } : {}),
|
||||
indices: [],
|
||||
};
|
||||
map.set(key, series);
|
||||
@@ -104,12 +88,11 @@ export function modelSeries(evaluations: readonly ModelRefLike[]): EvaluationSer
|
||||
return [...all.filter((x) => x.key !== ""), ...all.filter((x) => x.key === "")];
|
||||
}
|
||||
|
||||
/** A series' value sequence under the selected metric: indices outside this series are null (skipped point), keeping the global time axis. */
|
||||
/** A series' Score sequence; indices outside this series are null, keeping the global time axis. */
|
||||
export function seriesValues(
|
||||
evaluations: readonly (MetricSourceLike & ModelRefLike)[],
|
||||
series: EvaluationSeries,
|
||||
metric: BenchmarkMetric,
|
||||
): (number | null)[] {
|
||||
const own = new Set(series.indices);
|
||||
return metricValues(evaluations, metric).map((v, i) => (own.has(i) ? v : null));
|
||||
return scoreValues(evaluations).map((v, i) => (own.has(i) ? v : null));
|
||||
}
|
||||
|
||||
@@ -1,21 +1,18 @@
|
||||
/**
|
||||
* Benchmark page (read-only display):
|
||||
* the left directory lists Benchmarks grouped by Agent (the scoreboard is only fetched once
|
||||
* expanded); the right side shows the selected Benchmark's title info, a chart (switches between
|
||||
* score / cost / duration metrics on the same time axis; **grouped into series by the model each
|
||||
* evaluation carries** — the model isn't part of benchmark_config, each evaluation carries its
|
||||
* own — one color per series plus a legend, with older untagged records shown as a gray series;
|
||||
* missing values are skipped points, breaking the line; reuses the usage center's ChartFrame
|
||||
* coordinate system) and an evaluation detail table (includes a model column; rows expand to
|
||||
* show the evaluation summary — title and body shown separately — and per-case scores, and case
|
||||
* rows further expand to show the raw results of each run, with a Session link straight to that
|
||||
* Session's trace observability).
|
||||
* expanded); the right side shows the selected Benchmark's title info, a Score-only chart grouped
|
||||
* into series by each Evaluation's model ID and thinking level, and an evaluation detail table
|
||||
* with separate model ID and thinking-level columns. Rows expand to show the evaluation summary
|
||||
* and per-case scores, and Case rows further expand to show the raw results of each Run with a
|
||||
* Session link.
|
||||
* With a ?agentId= deep link, only the target Agent is expanded by default.
|
||||
*/
|
||||
import { useEffect, useState } from "react";
|
||||
import { Link, useSearchParams } from "react-router";
|
||||
import type {
|
||||
BenchmarkCaseScore,
|
||||
BenchmarkCaseSummary,
|
||||
BenchmarkEvaluation,
|
||||
BenchmarkSummary,
|
||||
} from "@prismshadow/penguin-server/api";
|
||||
@@ -29,22 +26,22 @@ import { useTheme } from "../../state/theme";
|
||||
import type { Currency } from "../../state/theme";
|
||||
import { AgentAvatar } from "../../components/ui/agent-avatar";
|
||||
import { Chevron } from "../../components/ui/chevron";
|
||||
import { Segmented } from "../../components/ui/segmented";
|
||||
import { Truncated } from "../../components/ui/truncated";
|
||||
import { EmptyState } from "../../components/ui/empty-state";
|
||||
import { Modal } from "../../components/ui/modal";
|
||||
import { SkeletonList } from "../../components/ui/skeleton";
|
||||
import { providerInfo } from "@prismshadow/penguin-core/model-catalog";
|
||||
import { seriesColor } from "../../lib/category-colors";
|
||||
import { makeGeom } from "../usage/chart-geom";
|
||||
import { ChartFrame, useChartWidth } from "../usage/chart-svg";
|
||||
import {
|
||||
lineSegments,
|
||||
metricMax,
|
||||
metricValues,
|
||||
modelSeries,
|
||||
scoreValues,
|
||||
seriesValues,
|
||||
} from "./benchmark-metrics";
|
||||
import type { BenchmarkMetric, EvaluationSeries } from "./benchmark-metrics";
|
||||
import type { EvaluationSeries } from "./benchmark-metrics";
|
||||
import { BenchmarkStatementBrowser } from "./benchmark-statement-browser";
|
||||
|
||||
interface Selection {
|
||||
agentId: string;
|
||||
@@ -144,61 +141,31 @@ function AgentNode({
|
||||
);
|
||||
}
|
||||
|
||||
/** Display label for a metric (shared by the Segmented options and the chart title; S is a live binding, must be read during render). */
|
||||
function metricLabel(metric: BenchmarkMetric): string {
|
||||
return metric === "score"
|
||||
? S.benchmark.colScore
|
||||
: metric === "cost"
|
||||
? S.common.cost
|
||||
: S.benchmark.colDuration;
|
||||
}
|
||||
|
||||
/** Display format for a metric value (shared by y-axis ticks and the tooltip): score / cost / duration each have their own formatting rule. */
|
||||
function formatMetric(metric: BenchmarkMetric, v: number, currency: Currency): string {
|
||||
return metric === "score"
|
||||
? formatScore(v)
|
||||
: metric === "cost"
|
||||
? formatMoney(v, currency)
|
||||
: humanizeDuration(v);
|
||||
}
|
||||
|
||||
/**
|
||||
* Metric-over-time line chart (cloned from the usage center's TrendChart: area + line + data
|
||||
* points, sharing the ChartFrame coordinate system). **Grouped into series** by the model each
|
||||
* evaluation carries: one color per series (SERIES_COLORS is a fixed color sequence, color
|
||||
* follows the model; older records with no model tag get a gray series), all series share the
|
||||
* same time axis and y-axis range. Missing values within a series (indices outside this series,
|
||||
* or cost / durationMs not recorded) are **skipped points** — lineSegments splits the
|
||||
* value-bearing indices into segments, drawing area + line + data points within each segment and
|
||||
* breaking between segments (a single-point segment draws only a point). ChartFrame's x-axis
|
||||
* labels take slice(5) of dates: passing `yyyy-MM-dd HH:mm` displays as `MM-dd HH:mm`. A single
|
||||
* evaluation is still drawn; no evaluations falls back to an empty state.
|
||||
* Score-over-time line chart. Every Run, Case, and Evaluation uses the fixed 0..100 scale.
|
||||
* Evaluations remain grouped by model ID and thinking level so a runtime change stays visible
|
||||
* without adding other metric modes.
|
||||
*/
|
||||
function MetricTrendChart({
|
||||
function ScoreTrendChart({
|
||||
evaluations,
|
||||
series,
|
||||
metric,
|
||||
currency,
|
||||
}: {
|
||||
evaluations: BenchmarkEvaluation[];
|
||||
series: EvaluationSeries[];
|
||||
metric: BenchmarkMetric;
|
||||
currency: Currency;
|
||||
}) {
|
||||
const [hover, setHover] = useState<number | null>(null);
|
||||
const [ref, width] = useChartWidth();
|
||||
|
||||
const values = metricValues(evaluations, metric);
|
||||
const geom = makeGeom(evaluations.length, metricMax(values), width);
|
||||
const values = scoreValues(evaluations);
|
||||
const geom = makeGeom(evaluations.length, Math.max(metricMax(values), 100), width);
|
||||
const dates = evaluations.map((e) => formatDateTime(e.time));
|
||||
const baseY = geom.y(0);
|
||||
|
||||
return (
|
||||
<div ref={ref}>
|
||||
{width > 0 && (
|
||||
<ChartFrame
|
||||
geom={geom}
|
||||
fmtY={(v) => formatMetric(metric, v, currency)}
|
||||
fmtY={formatScore}
|
||||
dates={dates}
|
||||
hover={hover}
|
||||
onHover={setHover}
|
||||
@@ -209,18 +176,20 @@ function MetricTrendChart({
|
||||
<>
|
||||
<p className="text-gray-400">{formatDateTime(e.time)}</p>
|
||||
<p className="font-mono">
|
||||
{v === null ? "—" : formatMetric(metric, v, currency)}
|
||||
{v === null ? "—" : formatScore(v)}
|
||||
{e.version !== undefined && (
|
||||
<span className="ml-1.5 text-gray-400">v{e.version}</span>
|
||||
)}
|
||||
</p>
|
||||
{e.modelId && <p className="font-mono text-gray-400">{e.modelId}</p>}
|
||||
<p className="font-mono text-gray-400">
|
||||
{e.modelId} · {e.thinkingLevel}
|
||||
</p>
|
||||
</>
|
||||
);
|
||||
}}
|
||||
>
|
||||
{series.map((s, si) => {
|
||||
const segments = lineSegments(seriesValues(evaluations, s, metric));
|
||||
const segments = lineSegments(seriesValues(evaluations, s));
|
||||
return (
|
||||
<g
|
||||
key={s.key === "" ? "unlabeled" : s.key}
|
||||
@@ -230,18 +199,8 @@ function MetricTrendChart({
|
||||
const line = seg
|
||||
.map((p, j) => `${j === 0 ? "M" : "L"}${geom.x(p.index)},${geom.y(p.value)}`)
|
||||
.join(" ");
|
||||
const area = `${line} L${geom.x(seg[seg.length - 1]!.index)},${baseY} L${geom.x(seg[0]!.index)},${baseY} Z`;
|
||||
return (
|
||||
<g key={k}>
|
||||
{/* Area fill: line closed to the baseline, low opacity reinforces the trend's sense of "volume" (no area for single-point segments) */}
|
||||
{seg.length > 1 && (
|
||||
<path
|
||||
d={area}
|
||||
className="fill-current"
|
||||
stroke="none"
|
||||
opacity={hover !== null ? 0.06 : 0.1}
|
||||
/>
|
||||
)}
|
||||
{seg.length > 1 && (
|
||||
<path
|
||||
d={line}
|
||||
@@ -274,55 +233,25 @@ function MetricTrendChart({
|
||||
}
|
||||
|
||||
/**
|
||||
* Chart section: title (follows metric) + metric switch (score / cost / duration, segmented
|
||||
* control) + model legend (only shown with >=2 series; a single series' identity is carried by
|
||||
* the detail table's model column and the hover tooltip instead) + the line chart. When the same
|
||||
* modelId coexists across providers, the legend appends the provider's display name to
|
||||
* disambiguate. Mounted under a keyed container per Benchmark: switching Benchmarks resets back
|
||||
* to "score".
|
||||
* Score chart + runtime legend. Provider is deliberately not part of chart identity.
|
||||
*/
|
||||
function TrendSection({
|
||||
evaluations,
|
||||
currency,
|
||||
}: {
|
||||
evaluations: BenchmarkEvaluation[];
|
||||
currency: Currency;
|
||||
}) {
|
||||
const [metric, setMetric] = useState<BenchmarkMetric>("score");
|
||||
function TrendSection({ evaluations }: { evaluations: BenchmarkEvaluation[] }) {
|
||||
const series = modelSeries(evaluations);
|
||||
const ids = series.map((s) => s.modelId).filter((v): v is string => v !== undefined);
|
||||
const dupIds = new Set(ids.filter((id, i) => ids.indexOf(id) !== i));
|
||||
const labelOf = (s: EvaluationSeries): string => {
|
||||
if (!s.modelId) return S.benchmark.legendUnlabeled;
|
||||
if (!dupIds.has(s.modelId)) return s.modelId;
|
||||
const provider = s.provider ? (providerInfo(s.provider)?.label ?? s.provider) : "";
|
||||
return provider ? `${s.modelId} · ${provider}` : s.modelId;
|
||||
return s.thinkingLevel ? `${s.modelId} · ${s.thinkingLevel}` : s.modelId;
|
||||
};
|
||||
return (
|
||||
<div>
|
||||
<div className="mb-1 flex items-center justify-between gap-2">
|
||||
<p className="text-xs font-semibold text-gray-500">
|
||||
{S.benchmark.trendTitle(metricLabel(metric))}
|
||||
</p>
|
||||
<div className="w-44 shrink-0">
|
||||
<Segmented
|
||||
options={[
|
||||
{ value: "score", label: metricLabel("score") },
|
||||
{ value: "cost", label: metricLabel("cost") },
|
||||
{ value: "duration", label: metricLabel("duration") },
|
||||
]}
|
||||
value={metric}
|
||||
onChange={setMetric}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
<p className="mb-1 text-xs font-semibold text-gray-500">
|
||||
{S.benchmark.trendTitle(S.benchmark.colScore)}
|
||||
</p>
|
||||
{series.length >= 2 && (
|
||||
<div className="mb-1.5 flex flex-wrap items-center gap-x-3 gap-y-1">
|
||||
{series.map((s, i) => (
|
||||
<span
|
||||
key={s.key === "" ? "unlabeled" : s.key}
|
||||
className="flex items-center gap-1.5 text-[11px] text-gray-500 dark:text-gray-400"
|
||||
title={s.provider ? (providerInfo(s.provider)?.label ?? s.provider) : undefined}
|
||||
>
|
||||
<span
|
||||
className={`inline-block h-2 w-2 shrink-0 rounded-sm ${
|
||||
@@ -334,26 +263,25 @@ function TrendSection({
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
<MetricTrendChart
|
||||
evaluations={evaluations}
|
||||
series={series}
|
||||
metric={metric}
|
||||
currency={currency}
|
||||
/>
|
||||
<ScoreTrendChart evaluations={evaluations} series={series} />
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
const CELL = "px-3 py-2";
|
||||
|
||||
/** One evaluation record: main row (time/version/total score/cost/duration) + a sub-table of per-case scores that expands on click. */
|
||||
/** One evaluation record: main row + a sub-table of per-Case scores that expands on click. */
|
||||
function EvaluationRow({
|
||||
agentId,
|
||||
evaluation,
|
||||
caseTitles,
|
||||
onOpenCase,
|
||||
currency,
|
||||
}: {
|
||||
agentId: string;
|
||||
evaluation: BenchmarkEvaluation;
|
||||
caseTitles: ReadonlyMap<string, string>;
|
||||
onOpenCase: (caseId: string) => void;
|
||||
currency: Currency;
|
||||
}) {
|
||||
const [open, setOpen] = useState(false);
|
||||
@@ -376,7 +304,10 @@ function EvaluationRow({
|
||||
className={`${CELL} max-w-40 truncate font-mono text-xs text-gray-500 dark:text-gray-400`}
|
||||
title={evaluation.provider}
|
||||
>
|
||||
{evaluation.modelId ?? "—"}
|
||||
{evaluation.modelId}
|
||||
</td>
|
||||
<td className={`${CELL} font-mono text-xs text-gray-500 dark:text-gray-400`}>
|
||||
{evaluation.thinkingLevel}
|
||||
</td>
|
||||
<td className={`${CELL} font-mono text-xs font-semibold tabular-nums`}>
|
||||
{formatScore(evaluation.score)}
|
||||
@@ -390,11 +321,8 @@ function EvaluationRow({
|
||||
</tr>
|
||||
{open && (
|
||||
<tr className="border-b border-gray-100 last:border-b-0 dark:border-gray-800/60">
|
||||
<td colSpan={6} className="bg-gray-50/80 px-3 py-2 dark:bg-gray-950/40">
|
||||
{/* Evaluation summary (title + body shown separately; the generating side always
|
||||
writes both, but the display side tolerates missing values — with an old-style
|
||||
single-paragraph summary only, it's still shown as usual, prefixed with the
|
||||
"Evaluation Summary" label). */}
|
||||
<td colSpan={7} className="bg-gray-50/80 px-3 py-2 dark:bg-gray-950/40">
|
||||
{/* Evaluation summary title and body are displayed separately when present. */}
|
||||
{(evaluation.summaryTitle || evaluation.summary) && (
|
||||
<div className="mb-2">
|
||||
{evaluation.summaryTitle ? (
|
||||
@@ -423,7 +351,14 @@ function EvaluationRow({
|
||||
</thead>
|
||||
<tbody>
|
||||
{evaluation.cases.map((c) => (
|
||||
<CaseRow key={c.case} agentId={agentId} caseScore={c} currency={currency} />
|
||||
<CaseRow
|
||||
key={c.case}
|
||||
agentId={agentId}
|
||||
caseScore={c}
|
||||
title={caseTitles.get(c.case)}
|
||||
onOpenCase={caseTitles.has(c.case) ? onOpenCase : undefined}
|
||||
currency={currency}
|
||||
/>
|
||||
))}
|
||||
</tbody>
|
||||
</table>
|
||||
@@ -449,33 +384,54 @@ function SessionLink({ agentId, sessionId }: { agentId: string; sessionId?: stri
|
||||
}
|
||||
|
||||
/**
|
||||
* Score row for one case: the case-level metrics = the average of its runs (already computed by
|
||||
* the server, trust its values). With runs[] present, the row can expand to show the raw results
|
||||
* of each run (#index + score / cost / duration / Session link); with the old format lacking
|
||||
* runs, it's not expandable and the case-level single Session link is used as before.
|
||||
* Score row for one Case: stored Case averages are authoritative. Expanding shows raw Run
|
||||
* results; the UI never recomputes averages.
|
||||
*/
|
||||
function CaseRow({
|
||||
agentId,
|
||||
caseScore: c,
|
||||
title,
|
||||
onOpenCase,
|
||||
currency,
|
||||
}: {
|
||||
agentId: string;
|
||||
caseScore: BenchmarkCaseScore;
|
||||
title?: string;
|
||||
onOpenCase?: (caseId: string) => void;
|
||||
currency: Currency;
|
||||
}) {
|
||||
const [open, setOpen] = useState(false);
|
||||
const runs = c.runs ?? [];
|
||||
const expandable = runs.length > 0;
|
||||
const runs = c.runs;
|
||||
return (
|
||||
<>
|
||||
<tr
|
||||
onClick={expandable ? () => setOpen((v) => !v) : undefined}
|
||||
className={`text-xs ${expandable ? "cursor-pointer transition-colors duration-150 hover:bg-gray-100/70 dark:hover:bg-gray-800/40" : ""}`}
|
||||
onClick={() => setOpen((v) => !v)}
|
||||
className="cursor-pointer text-xs transition-colors duration-150 hover:bg-gray-100/70 dark:hover:bg-gray-800/40"
|
||||
>
|
||||
<td className="px-2 py-1 font-mono">
|
||||
<span className="flex items-center gap-1.5">
|
||||
{expandable && <Chevron open={open} size={12} className="text-gray-400" />}
|
||||
{c.case}
|
||||
<td className="px-2 py-1">
|
||||
<span className="flex items-start gap-1.5">
|
||||
<Chevron open={open} size={12} className="text-gray-400" />
|
||||
<span className="min-w-0">
|
||||
{onOpenCase ? (
|
||||
<button
|
||||
type="button"
|
||||
className="block text-left font-medium text-gray-800 hover:underline dark:text-gray-200"
|
||||
onClick={(event) => {
|
||||
event.stopPropagation();
|
||||
onOpenCase(c.case);
|
||||
}}
|
||||
>
|
||||
{title ?? c.case}
|
||||
</button>
|
||||
) : (
|
||||
<span className="block font-medium text-gray-800 dark:text-gray-200">
|
||||
{title ?? c.case}
|
||||
</span>
|
||||
)}
|
||||
{title && title !== c.case && (
|
||||
<span className="block font-mono text-[11px] text-gray-400">{c.case}</span>
|
||||
)}
|
||||
</span>
|
||||
</span>
|
||||
</td>
|
||||
<td className="px-2 py-1 font-mono tabular-nums">{formatScore(c.score)}</td>
|
||||
@@ -486,7 +442,7 @@ function CaseRow({
|
||||
{c.durationMs !== undefined ? humanizeDuration(c.durationMs) : "—"}
|
||||
</td>
|
||||
<td className="px-2 py-1">
|
||||
<SessionLink agentId={agentId} {...(c.sessionId ? { sessionId: c.sessionId } : {})} />
|
||||
<span className="text-gray-400">—</span>
|
||||
</td>
|
||||
</tr>
|
||||
{open &&
|
||||
@@ -513,6 +469,51 @@ function CaseRow({
|
||||
);
|
||||
}
|
||||
|
||||
function CasesSection({
|
||||
cases,
|
||||
error,
|
||||
onOpenCase,
|
||||
}: {
|
||||
cases: BenchmarkCaseSummary[] | null;
|
||||
error: string | null;
|
||||
onOpenCase: (caseId: string) => void;
|
||||
}) {
|
||||
return (
|
||||
<div>
|
||||
<p className="mb-1 text-xs font-semibold text-gray-500">{S.benchmark.cases}</p>
|
||||
<div className="overflow-hidden rounded-md border border-gray-200 bg-white dark:border-gray-800 dark:bg-gray-900">
|
||||
{error && <p className="px-3 py-2 text-xs text-red-500">{error}</p>}
|
||||
{!cases && !error && <p className="px-3 py-2 text-xs text-gray-400">{S.common.loading}</p>}
|
||||
{cases?.map((item) => {
|
||||
return (
|
||||
<button
|
||||
key={item.id}
|
||||
type="button"
|
||||
onClick={() => onOpenCase(item.id)}
|
||||
className="flex w-full items-center gap-3 border-b border-gray-100 px-3 py-2 text-left transition-colors last:border-b-0 hover:bg-gray-50 dark:border-gray-800/70 dark:hover:bg-gray-800/50"
|
||||
>
|
||||
<span className="min-w-0 flex-1">
|
||||
<span className="block text-sm font-medium text-gray-800 dark:text-gray-200">
|
||||
{item.title}
|
||||
</span>
|
||||
<span className="block truncate font-mono text-[11px] text-gray-400">
|
||||
{item.id}
|
||||
</span>
|
||||
</span>
|
||||
<span className="shrink-0 font-mono text-xs text-gray-500">
|
||||
{S.benchmark.maxScore("100")}
|
||||
</span>
|
||||
<span className="shrink-0 text-xs text-brand-700 dark:text-brand-300">
|
||||
{S.benchmark.viewCase}
|
||||
</span>
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
export function BenchmarkPage() {
|
||||
useDocumentTitle(S.benchmark.title);
|
||||
const { currentProject, agents, agentsLoading } = useProject();
|
||||
@@ -522,17 +523,42 @@ export function BenchmarkPage() {
|
||||
const [searchParams] = useSearchParams();
|
||||
const focusAgentId = searchParams.get("agentId");
|
||||
const [selection, setSelection] = useState<Selection | null>(null);
|
||||
const [caseStatements, setCaseStatements] = useState<BenchmarkCaseSummary[] | null>(null);
|
||||
const [caseError, setCaseError] = useState<string | null>(null);
|
||||
const [openCaseId, setOpenCaseId] = useState<string | null>(null);
|
||||
|
||||
// Clear the selection when the Project changes.
|
||||
useEffect(() => {
|
||||
setSelection(null);
|
||||
}, [projectId]);
|
||||
|
||||
useEffect(() => {
|
||||
setCaseStatements(null);
|
||||
setCaseError(null);
|
||||
setOpenCaseId(null);
|
||||
if (!projectId || !selection) return;
|
||||
let cancelled = false;
|
||||
api
|
||||
.listBenchmarkCases(projectId, selection.agentId, selection.benchmark.id)
|
||||
.then((data) => {
|
||||
if (!cancelled) setCaseStatements(data.cases);
|
||||
})
|
||||
.catch((error: unknown) => {
|
||||
if (!cancelled) setCaseError(apiErrorText(error));
|
||||
});
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, [projectId, selection]);
|
||||
|
||||
if (!projectId) return null;
|
||||
|
||||
const bm = selection?.benchmark ?? null;
|
||||
// Chart uses ascending time order (the scoreboard is already ordered, this sort is defensive); the detail table shows newest first.
|
||||
const evaluations = bm ? [...bm.evaluations].sort((a, b) => a.time.localeCompare(b.time)) : [];
|
||||
// The Scoreboard append order is the evaluation sequence. Preserve it even when a malformed
|
||||
// timestamp would otherwise reorder Agent versions; the detail table shows that sequence newest first.
|
||||
const evaluations = bm ? [...bm.evaluations] : [];
|
||||
const caseTitles = new Map(caseStatements?.map((item) => [item.id, item.title]) ?? []);
|
||||
const openCase = caseStatements?.find((item) => item.id === openCaseId) ?? null;
|
||||
|
||||
return (
|
||||
<div className="flex h-full flex-col md:flex-row">
|
||||
@@ -564,9 +590,7 @@ export function BenchmarkPage() {
|
||||
{selection && bm ? (
|
||||
// Changing the key on Benchmark switch resets expand state (a detail row's open doesn't linger across Benchmarks).
|
||||
<div key={`${selection.agentId}/${bm.id}`} className="mx-auto max-w-4xl space-y-4">
|
||||
{/* Title row: title + case count (the model isn't part of config — each evaluation
|
||||
carries its own, see the chart legend and the detail table's model column) +
|
||||
description */}
|
||||
{/* Runtime belongs to each Evaluation and is shown in the detail table. */}
|
||||
<div>
|
||||
<div className="flex flex-wrap items-baseline gap-x-2 gap-y-1">
|
||||
<h1 className="min-w-0 truncate text-lg font-semibold">{bm.title}</h1>
|
||||
@@ -577,23 +601,26 @@ export function BenchmarkPage() {
|
||||
)}
|
||||
</div>
|
||||
|
||||
<CasesSection cases={caseStatements} error={caseError} onOpenCase={setOpenCaseId} />
|
||||
|
||||
{evaluations.length === 0 ? (
|
||||
<EmptyState title={S.benchmark.noEvaluations} />
|
||||
) : (
|
||||
<>
|
||||
<TrendSection evaluations={evaluations} currency={currency} />
|
||||
<TrendSection evaluations={evaluations} />
|
||||
|
||||
<div>
|
||||
<p className="mb-1 text-xs font-semibold text-gray-500">
|
||||
{S.benchmark.evaluations}
|
||||
</p>
|
||||
<div className="overflow-x-auto overflow-y-clip rounded-md border border-gray-200 bg-white dark:border-gray-800 dark:bg-gray-900">
|
||||
<table className="w-full min-w-[600px] text-left text-sm">
|
||||
<table className="w-full min-w-[720px] text-left text-sm">
|
||||
<thead>
|
||||
<tr className="border-b border-gray-200 bg-gray-50/80 text-xs text-gray-500 dark:border-gray-800 dark:bg-gray-900">
|
||||
<th className="px-3 py-2.5">{S.common.time}</th>
|
||||
<th className="px-3 py-2.5">{S.benchmark.colVersion}</th>
|
||||
<th className="px-3 py-2.5">{S.benchmark.colModel}</th>
|
||||
<th className="px-3 py-2.5">{S.benchmark.colThinkingLevel}</th>
|
||||
<th className="px-3 py-2.5">{S.benchmark.colScore}</th>
|
||||
<th className="px-3 py-2.5">{S.common.cost}</th>
|
||||
<th className="px-3 py-2.5">{S.benchmark.colDuration}</th>
|
||||
@@ -605,6 +632,8 @@ export function BenchmarkPage() {
|
||||
key={i}
|
||||
agentId={selection.agentId}
|
||||
evaluation={ev}
|
||||
caseTitles={caseTitles}
|
||||
onOpenCase={setOpenCaseId}
|
||||
currency={currency}
|
||||
/>
|
||||
))}
|
||||
@@ -614,6 +643,21 @@ export function BenchmarkPage() {
|
||||
</div>
|
||||
</>
|
||||
)}
|
||||
{openCase && (
|
||||
<Modal
|
||||
open
|
||||
title={openCase.title}
|
||||
widthClass="sm:max-w-6xl"
|
||||
onClose={() => setOpenCaseId(null)}
|
||||
>
|
||||
<BenchmarkStatementBrowser
|
||||
projectId={projectId}
|
||||
agentId={selection.agentId}
|
||||
benchmarkId={bm.id}
|
||||
caseSummary={openCase}
|
||||
/>
|
||||
</Modal>
|
||||
)}
|
||||
</div>
|
||||
) : (
|
||||
<EmptyState title={S.benchmark.selectBenchmark} />
|
||||
|
||||
@@ -0,0 +1,367 @@
|
||||
import { useCallback, useEffect, useRef, useState } from "react";
|
||||
import type {
|
||||
BenchmarkCaseSummary,
|
||||
WorkspaceFileEntry,
|
||||
WorkspaceFilesResponse,
|
||||
} from "@prismshadow/penguin-server/api";
|
||||
import ReactMarkdown from "react-markdown";
|
||||
import type { Components } from "react-markdown";
|
||||
import remarkGfm from "remark-gfm";
|
||||
import * as api from "../../api/endpoints";
|
||||
import { apiErrorText } from "../../lib/api-error";
|
||||
import { formatBytes } from "../../lib/format";
|
||||
import { S } from "../../lib/strings";
|
||||
import { SkeletonList } from "../../components/ui/skeleton";
|
||||
import { CodeBlock } from "../chat/code-block";
|
||||
|
||||
const TEXT_EXTS = new Set([
|
||||
"txt",
|
||||
"md",
|
||||
"json",
|
||||
"js",
|
||||
"mjs",
|
||||
"cjs",
|
||||
"ts",
|
||||
"tsx",
|
||||
"jsx",
|
||||
"py",
|
||||
"sh",
|
||||
"bash",
|
||||
"yaml",
|
||||
"yml",
|
||||
"toml",
|
||||
"css",
|
||||
"html",
|
||||
"htm",
|
||||
"csv",
|
||||
"log",
|
||||
"xml",
|
||||
"ini",
|
||||
"conf",
|
||||
"sql",
|
||||
"svg",
|
||||
]);
|
||||
const IMAGE_EXTS = new Set(["png", "jpg", "jpeg", "gif", "webp"]);
|
||||
const EXTERNAL_REF_RE = /^[a-z][a-z0-9+.-]*:/i;
|
||||
const HIGHLIGHT_LIMIT = 64 * 1024;
|
||||
|
||||
interface Preview {
|
||||
path: string;
|
||||
name: string;
|
||||
kind: "text" | "md" | "image" | "pdf" | "unsupported";
|
||||
content?: string;
|
||||
truncated?: boolean;
|
||||
loading?: boolean;
|
||||
error?: string;
|
||||
}
|
||||
|
||||
interface Props {
|
||||
projectId: string;
|
||||
agentId: string;
|
||||
benchmarkId: string;
|
||||
caseSummary: BenchmarkCaseSummary;
|
||||
}
|
||||
|
||||
function extOf(name: string): string {
|
||||
const index = name.lastIndexOf(".");
|
||||
return index >= 0 ? name.slice(index + 1).toLowerCase() : name.toLowerCase();
|
||||
}
|
||||
|
||||
function joinPath(dir: string, name: string): string {
|
||||
return dir === "" ? name : `${dir}/${name}`;
|
||||
}
|
||||
|
||||
function dirOf(filePath: string): string {
|
||||
return filePath.includes("/") ? filePath.slice(0, filePath.lastIndexOf("/")) : "";
|
||||
}
|
||||
|
||||
function resolveRelative(baseDir: string, ref: string): string {
|
||||
const out = ref.startsWith("/") || baseDir === "" ? [] : baseDir.split("/");
|
||||
for (const segment of ref.split("/")) {
|
||||
if (segment === "" || segment === ".") continue;
|
||||
if (segment === "..") out.pop();
|
||||
else out.push(segment);
|
||||
}
|
||||
return out.join("/");
|
||||
}
|
||||
|
||||
function languageFor(name: string): string {
|
||||
const ext = extOf(name);
|
||||
return (
|
||||
{
|
||||
json: "json",
|
||||
md: "markdown",
|
||||
js: "javascript",
|
||||
mjs: "javascript",
|
||||
cjs: "javascript",
|
||||
jsx: "jsx",
|
||||
ts: "typescript",
|
||||
tsx: "tsx",
|
||||
py: "python",
|
||||
sh: "shellscript",
|
||||
bash: "shellscript",
|
||||
yaml: "yaml",
|
||||
yml: "yaml",
|
||||
toml: "toml",
|
||||
css: "css",
|
||||
html: "html",
|
||||
htm: "html",
|
||||
xml: "xml",
|
||||
sql: "sql",
|
||||
svg: "xml",
|
||||
}[ext] ?? "text"
|
||||
);
|
||||
}
|
||||
|
||||
export function BenchmarkStatementBrowser({ projectId, agentId, benchmarkId, caseSummary }: Props) {
|
||||
const [path, setPath] = useState("");
|
||||
const [listing, setListing] = useState<WorkspaceFilesResponse | null>(null);
|
||||
const [listError, setListError] = useState<string | null>(null);
|
||||
const [preview, setPreview] = useState<Preview | null>(null);
|
||||
const initialReadmeOpened = useRef(false);
|
||||
const previewRequest = useRef(0);
|
||||
|
||||
const fileUrl = useCallback(
|
||||
(filePath: string, options?: { download?: boolean; preview?: boolean }) =>
|
||||
api.benchmarkCaseFileUrl(projectId, agentId, benchmarkId, caseSummary.id, filePath, options),
|
||||
[projectId, agentId, benchmarkId, caseSummary.id],
|
||||
);
|
||||
|
||||
const previewPath = useCallback(
|
||||
async (filePath: string) => {
|
||||
const request = ++previewRequest.current;
|
||||
const name = filePath.includes("/")
|
||||
? filePath.slice(filePath.lastIndexOf("/") + 1)
|
||||
: filePath;
|
||||
const ext = extOf(name);
|
||||
if (IMAGE_EXTS.has(ext)) {
|
||||
setPreview({ path: filePath, name, kind: "image" });
|
||||
return;
|
||||
}
|
||||
if (ext === "pdf") {
|
||||
setPreview({ path: filePath, name, kind: "pdf" });
|
||||
return;
|
||||
}
|
||||
const isMarkdown = ext === "md";
|
||||
if (!TEXT_EXTS.has(ext)) {
|
||||
setPreview({ path: filePath, name, kind: "unsupported" });
|
||||
return;
|
||||
}
|
||||
setPreview({ path: filePath, name, kind: isMarkdown ? "md" : "text", loading: true });
|
||||
try {
|
||||
const response = await fetch(fileUrl(filePath, { preview: true }), {
|
||||
credentials: "same-origin",
|
||||
});
|
||||
if (!response.ok) throw new Error(String(response.status));
|
||||
const content = await response.text();
|
||||
if (request !== previewRequest.current) return;
|
||||
setPreview({
|
||||
path: filePath,
|
||||
name,
|
||||
kind: isMarkdown ? "md" : "text",
|
||||
content,
|
||||
truncated: response.headers.get("x-content-truncated") === "1",
|
||||
});
|
||||
} catch (error) {
|
||||
if (request !== previewRequest.current) return;
|
||||
setPreview({
|
||||
path: filePath,
|
||||
name,
|
||||
kind: isMarkdown ? "md" : "text",
|
||||
error: apiErrorText(error),
|
||||
});
|
||||
}
|
||||
},
|
||||
[fileUrl],
|
||||
);
|
||||
|
||||
useEffect(() => {
|
||||
setListing(null);
|
||||
setListError(null);
|
||||
let cancelled = false;
|
||||
api
|
||||
.listBenchmarkCaseFiles(projectId, agentId, benchmarkId, caseSummary.id, path)
|
||||
.then((data) => {
|
||||
if (cancelled) return;
|
||||
setListing(data);
|
||||
if (path === "" && !initialReadmeOpened.current) {
|
||||
initialReadmeOpened.current = true;
|
||||
const readme = data.entries.find(
|
||||
(entry) => entry.kind === "file" && entry.name.toLowerCase() === "readme.md",
|
||||
);
|
||||
if (readme) void previewPath(readme.name);
|
||||
}
|
||||
})
|
||||
.catch((error: unknown) => {
|
||||
if (!cancelled) setListError(apiErrorText(error));
|
||||
});
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, [projectId, agentId, benchmarkId, caseSummary.id, path, previewPath]);
|
||||
|
||||
const crumbs = path === "" ? [] : path.split("/");
|
||||
const downloadUrl = preview ? fileUrl(preview.path, { download: true }) : null;
|
||||
|
||||
const openEntry = (entry: WorkspaceFileEntry) => {
|
||||
if (entry.kind === "dir") {
|
||||
setPath(joinPath(path, entry.name));
|
||||
return;
|
||||
}
|
||||
void previewPath(joinPath(path, entry.name));
|
||||
};
|
||||
|
||||
const markdownComponents: Components = {
|
||||
img: ({ src, alt }) => (
|
||||
<img
|
||||
src={
|
||||
typeof src === "string" && !EXTERNAL_REF_RE.test(src)
|
||||
? fileUrl(resolveRelative(dirOf(preview?.path ?? ""), src))
|
||||
: src
|
||||
}
|
||||
alt={alt ?? ""}
|
||||
loading="lazy"
|
||||
className="max-w-full"
|
||||
/>
|
||||
),
|
||||
a: ({ href, children }) => {
|
||||
if (typeof href !== "string" || href.startsWith("#")) return <a href={href}>{children}</a>;
|
||||
if (EXTERNAL_REF_RE.test(href)) {
|
||||
return (
|
||||
<a href={href} target="_blank" rel="noreferrer">
|
||||
{children}
|
||||
</a>
|
||||
);
|
||||
}
|
||||
const target = resolveRelative(dirOf(preview?.path ?? ""), href);
|
||||
return (
|
||||
<a
|
||||
href={fileUrl(target)}
|
||||
onClick={(event) => {
|
||||
event.preventDefault();
|
||||
void previewPath(target);
|
||||
}}
|
||||
>
|
||||
{children}
|
||||
</a>
|
||||
);
|
||||
},
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="grid min-h-[58vh] overflow-hidden rounded-md border border-gray-200 md:grid-cols-[240px_minmax(0,1fr)] dark:border-gray-800">
|
||||
<aside className="border-b border-gray-200 bg-gray-50/60 md:border-b-0 md:border-r dark:border-gray-800 dark:bg-gray-950/30">
|
||||
<div className="flex flex-wrap items-center gap-1 border-b border-gray-200 px-2 py-2 dark:border-gray-800">
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => setPath("")}
|
||||
className="rounded px-1.5 py-0.5 text-xs text-gray-600 hover:bg-gray-100 dark:text-gray-300 dark:hover:bg-gray-800"
|
||||
>
|
||||
{S.benchmark.publicMaterials}
|
||||
</button>
|
||||
{crumbs.map((segment, index) => (
|
||||
<span key={`${segment}-${index}`} className="flex min-w-0 items-center gap-1">
|
||||
<span className="text-gray-300 dark:text-gray-700">/</span>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => setPath(crumbs.slice(0, index + 1).join("/"))}
|
||||
className="max-w-24 truncate rounded px-1 py-0.5 text-xs text-gray-600 hover:bg-gray-100 dark:text-gray-300 dark:hover:bg-gray-800"
|
||||
>
|
||||
{segment}
|
||||
</button>
|
||||
</span>
|
||||
))}
|
||||
</div>
|
||||
<div className="max-h-44 overflow-y-auto md:max-h-[53vh]">
|
||||
{listError && <p className="px-3 py-2 text-xs text-red-500">{listError}</p>}
|
||||
{!listing && !listError && <SkeletonList rows={4} />}
|
||||
{listing?.entries.length === 0 && (
|
||||
<p className="px-3 py-2 text-xs text-gray-400">{S.files.empty}</p>
|
||||
)}
|
||||
{listing?.entries.map((entry) => (
|
||||
<button
|
||||
key={`${entry.kind}/${entry.name}`}
|
||||
type="button"
|
||||
onClick={() => openEntry(entry)}
|
||||
className="flex w-full items-center gap-2 border-b border-gray-100 px-3 py-2 text-left hover:bg-gray-100 dark:border-gray-800/70 dark:hover:bg-gray-800/60"
|
||||
>
|
||||
<span className="text-sm text-gray-400">{entry.kind === "dir" ? "▸" : "·"}</span>
|
||||
<span className="min-w-0 flex-1 truncate text-sm">{entry.name}</span>
|
||||
{entry.kind === "file" && (
|
||||
<span className="shrink-0 text-[11px] text-gray-400">
|
||||
{formatBytes(entry.sizeBytes)}
|
||||
</span>
|
||||
)}
|
||||
</button>
|
||||
))}
|
||||
</div>
|
||||
</aside>
|
||||
|
||||
<section className="min-w-0">
|
||||
<div className="flex min-h-11 flex-wrap items-center gap-2 border-b border-gray-200 px-3 py-2 dark:border-gray-800">
|
||||
<div className="min-w-0 flex-1">
|
||||
<p className="truncate font-mono text-xs text-gray-500">
|
||||
{preview?.path ?? caseSummary.id}
|
||||
</p>
|
||||
</div>
|
||||
<span className="shrink-0 text-xs text-gray-400">{S.benchmark.maxScore("100")}</span>
|
||||
{downloadUrl && preview && (
|
||||
<a
|
||||
href={downloadUrl}
|
||||
download={preview.name}
|
||||
className="rounded-md px-2.5 py-1 text-xs font-medium text-gray-600 hover:bg-gray-100 dark:text-gray-300 dark:hover:bg-gray-800"
|
||||
>
|
||||
{S.files.download}
|
||||
</a>
|
||||
)}
|
||||
</div>
|
||||
<div className="max-h-[52vh] min-h-[52vh] overflow-auto p-3">
|
||||
{!preview ? (
|
||||
<p className="text-sm text-gray-400">{S.benchmark.statementUnavailable}</p>
|
||||
) : preview.loading ? (
|
||||
<SkeletonList rows={8} />
|
||||
) : preview.error ? (
|
||||
<p className="text-sm text-red-500">{preview.error}</p>
|
||||
) : preview.kind === "image" ? (
|
||||
<img
|
||||
src={fileUrl(preview.path)}
|
||||
alt={preview.name}
|
||||
loading="lazy"
|
||||
className="max-w-full rounded-md border border-gray-200 dark:border-gray-800"
|
||||
/>
|
||||
) : preview.kind === "pdf" ? (
|
||||
<iframe
|
||||
src={fileUrl(preview.path)}
|
||||
title={preview.name}
|
||||
className="h-[50vh] w-full rounded-md border border-gray-200 dark:border-gray-800"
|
||||
/>
|
||||
) : preview.kind === "md" ? (
|
||||
<>
|
||||
<div className="md-body text-sm text-gray-800 dark:text-gray-100">
|
||||
<ReactMarkdown remarkPlugins={[remarkGfm]} components={markdownComponents}>
|
||||
{preview.content ?? ""}
|
||||
</ReactMarkdown>
|
||||
</div>
|
||||
{preview.truncated && (
|
||||
<p className="mt-2 text-xs text-gray-400">{S.files.previewTruncated}</p>
|
||||
)}
|
||||
</>
|
||||
) : preview.kind === "text" ? (
|
||||
<>
|
||||
<CodeBlock
|
||||
language={languageFor(preview.name)}
|
||||
code={preview.content ?? ""}
|
||||
highlight={(preview.content?.length ?? 0) <= HIGHLIGHT_LIMIT}
|
||||
/>
|
||||
{preview.truncated && (
|
||||
<p className="mt-2 text-xs text-gray-400">{S.files.previewTruncated}</p>
|
||||
)}
|
||||
</>
|
||||
) : (
|
||||
<p className="text-sm text-gray-500 dark:text-gray-400">{S.files.previewUnsupported}</p>
|
||||
)}
|
||||
</div>
|
||||
</section>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -55,6 +55,7 @@ import { toastError } from "../../components/ui/toast";
|
||||
import { useVersionInfo } from "../../lib/use-version-info";
|
||||
import { ChatInput } from "./chat-input";
|
||||
import { buildSkillsMessage } from "./skill-use";
|
||||
import { EXAMPLE_TASKS, type ExampleTask, type ExampleTaskId } from "./example-tasks";
|
||||
import { clearDraft, draftKey, loadDraft, saveDraft } from "./draft-cache";
|
||||
import type { DraftCache } from "./draft-cache";
|
||||
import { sameModelRef } from "../models/model-grouping";
|
||||
@@ -88,18 +89,6 @@ function saveAppliedRouteKey(field: RouteStateField, key: string): void {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Example tasks on the draft screen, in display order (game card first, LoL music player,
|
||||
* then the RAG build). Copy lives in S.chat.exampleTasks[id]; skills are pinned via a
|
||||
* `[use_skills]` block — only those the selected Agent actually has installed are included,
|
||||
* so the block never references a skill the agent can't read.
|
||||
*/
|
||||
const EXAMPLE_TASKS: { id: "game" | "lol" | "rag"; skills: string[] }[] = [
|
||||
{ id: "game", skills: ["web-design"] },
|
||||
{ id: "lol", skills: ["web-design"] },
|
||||
{ id: "rag", skills: ["penguin-sdk", "web-design"] },
|
||||
];
|
||||
|
||||
export function DraftView({
|
||||
projectId,
|
||||
models,
|
||||
@@ -415,9 +404,9 @@ export function DraftView({
|
||||
// busy id drives the clicked card's spinner; the shared in-flight guard and all failure
|
||||
// handling live in onSend). keepDraft: an example never consumes the composer text, so a
|
||||
// typed-but-unsent draft must survive. The selected model / Workspace / approval mode apply as-is.
|
||||
const [exampleBusy, setExampleBusy] = useState<"game" | "lol" | "rag" | null>(null);
|
||||
const [exampleBusy, setExampleBusy] = useState<ExampleTaskId | null>(null);
|
||||
const runExample = useCallback(
|
||||
async (task: (typeof EXAMPLE_TASKS)[number]) => {
|
||||
async (task: ExampleTask) => {
|
||||
if (exampleBusy !== null) return;
|
||||
setExampleBusy(task.id);
|
||||
try {
|
||||
@@ -517,7 +506,25 @@ export function DraftView({
|
||||
className="group flex min-w-0 items-center gap-3 rounded-xl border border-gray-200 bg-white px-4 py-3 text-left transition-colors duration-150 hover:border-gray-300 disabled:cursor-default disabled:opacity-60 dark:border-gray-800 dark:bg-gray-900 dark:hover:border-gray-700"
|
||||
>
|
||||
{/* 24×24 line icons (gamepad / music note / sparkle), consistent with the icon convention */}
|
||||
{task.id === "lol" ? (
|
||||
{task.id === "agentBenchmarkBuild" || task.id === "agentOptimization" ? (
|
||||
<svg
|
||||
width="20"
|
||||
height="20"
|
||||
viewBox="0 0 24 24"
|
||||
fill="none"
|
||||
stroke="currentColor"
|
||||
className="shrink-0 text-brand-500 dark:text-brand-400"
|
||||
aria-hidden
|
||||
>
|
||||
<path
|
||||
d="M7 7h8a4 4 0 0 1 4 4v1M17 5l2 2-2 2M17 17H9a4 4 0 0 1-4-4v-1M7 19l-2-2 2-2"
|
||||
strokeWidth="1.7"
|
||||
strokeLinecap="round"
|
||||
strokeLinejoin="round"
|
||||
/>
|
||||
<circle cx="12" cy="12" r="2" strokeWidth="1.7" />
|
||||
</svg>
|
||||
) : task.id === "lol" ? (
|
||||
<svg
|
||||
width="20"
|
||||
height="20"
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
/**
|
||||
* Draft-screen example cards in display order.
|
||||
*
|
||||
* Copy and full prompts live in the active locale dictionary at
|
||||
* `S.chat.exampleTasks[id]`. Skills listed here are pinned only when the
|
||||
* selected Agent has them installed; an empty list sends the prompt unchanged.
|
||||
*/
|
||||
export const EXAMPLE_TASKS = [
|
||||
{ id: "game", skills: ["web-design"] },
|
||||
{ id: "lol", skills: ["web-design"] },
|
||||
{ id: "rag", skills: ["penguin-sdk", "web-design"] },
|
||||
{ id: "agentBenchmarkBuild", skills: [] },
|
||||
{ id: "agentOptimization", skills: [] },
|
||||
] as const;
|
||||
|
||||
export type ExampleTask = (typeof EXAMPLE_TASKS)[number];
|
||||
export type ExampleTaskId = ExampleTask["id"];
|
||||
@@ -39,8 +39,8 @@ export interface SeriesColor {
|
||||
}
|
||||
|
||||
/**
|
||||
* Fixed color sequence for line series (the eval center splits series by
|
||||
* model): maintained separately from CATEGORY_COLORS — adjacent line colors
|
||||
* Fixed color sequence for line series (the eval center splits series by model ID and thinking
|
||||
* level): maintained separately from CATEGORY_COLORS — adjacent line colors
|
||||
* must pass color-vision-deficiency (CVD) separation checks and dark shades
|
||||
* must land within the brightness band. The violet / amber / sky / rose
|
||||
* ordering passes the dataviz validator in both modes (dark-mode amber / sky
|
||||
|
||||
@@ -126,9 +126,9 @@ export function formatMoney(
|
||||
return `${symbol}${v.toFixed(digits)}`;
|
||||
}
|
||||
|
||||
/** Benchmark total score display: integers unchanged, decimals keep one place (full-score convention is defined per-Benchmark by its scoring rubric). */
|
||||
/** Benchmark score display: integers unchanged, decimals keep up to the stored two places. */
|
||||
export function formatScore(n: number): string {
|
||||
return Number.isInteger(n) ? `${n}` : trimZero(n);
|
||||
return Number.isInteger(n) ? `${n}` : n.toFixed(2).replace(/0+$/, "").replace(/\.$/, "");
|
||||
}
|
||||
|
||||
/** True while the one-decimal rendering of `v` still fits its unit: 1023.9 does, 1023.99 does not. */
|
||||
|
||||
@@ -604,6 +604,39 @@ When done, open index.html in a browser and self-test once.`,
|
||||
"give it a beautiful web chat UI following the web-design skill, with a few example questions in the empty state. " +
|
||||
"When done, run the app, verify one streamed answer yourself, and tell me how to access it.",
|
||||
},
|
||||
agentBenchmarkBuild: {
|
||||
label: "Example: create a decision Agent and capability evaluation",
|
||||
desc: "Create a general decision Agent and test it on football, after-sales, and investment tasks",
|
||||
prompt: `Use \`agent-creation\` followed by \`benchmark-design\` to create a decision Agent and produce a frozen Benchmark with a Formal Baseline.
|
||||
|
||||
Agent:
|
||||
- id: \`finite_choice_agent\`
|
||||
- capability: make stable, explainable finite choices when public information is incomplete or conflicting
|
||||
- installed_skills: \`[]\`
|
||||
|
||||
Benchmark:
|
||||
- id: \`contextual-choice-adaptation\`
|
||||
- capability: form and transfer a stable finite-choice decision process from public rules, historical examples, and current facts
|
||||
- runs: \`1\`
|
||||
- desired_baseline_score: \`<75\`
|
||||
- pilot_iteration_limit: \`5\`
|
||||
|
||||
Scenarios:
|
||||
1. Make football betting decisions from historical matches and current information.
|
||||
2. Choose after-sales actions from policy and ticket facts.
|
||||
3. Choose investment actions from a strategy, historical markets, and current indicators.`,
|
||||
},
|
||||
agentOptimization: {
|
||||
label: "Example: optimize a decision Agent from its evaluation",
|
||||
desc: "Improve an Agent from existing evaluation results and verify that the new version is better",
|
||||
prompt: `Use \`agent-optimization\` to optimize a decision Agent against its frozen Benchmark.
|
||||
|
||||
- test_agent_id: \`finite_choice_agent\`
|
||||
- benchmark_id: \`contextual-choice-adaptation\`
|
||||
- capability_direction: improve stability under incomplete information, conflicting rules, and finite choices
|
||||
- desired_score: \`>=95\`
|
||||
- candidate_round_limit: \`5\``,
|
||||
},
|
||||
},
|
||||
sessionList: "Sessions",
|
||||
defaultSessionTitle: "New chat",
|
||||
@@ -944,12 +977,18 @@ When done, open index.html in a browser and self-test once.`,
|
||||
emptyAgent: "No Benchmarks for this agent",
|
||||
caseCount: (n: number): string => `${n} case${n === 1 ? "" : "s"}`,
|
||||
trendTitle: (metric: string): string => `${metric} over time`,
|
||||
cases: "Cases",
|
||||
viewCase: "View task",
|
||||
publicMaterials: "Public materials",
|
||||
statementUnavailable: "The public Statement is unavailable",
|
||||
maxScore: (score: string): string => `Max ${score}`,
|
||||
evaluations: "Evaluations",
|
||||
noEvaluations: "No evaluations yet",
|
||||
summaryLabel: "Summary",
|
||||
legendUnlabeled: "unlabeled model",
|
||||
colVersion: "Version",
|
||||
colModel: "Model",
|
||||
colModel: "Model ID",
|
||||
colThinkingLevel: "Thinking level",
|
||||
colScore: "Score",
|
||||
colDuration: "Duration",
|
||||
colCase: "Case",
|
||||
|
||||
@@ -532,11 +532,9 @@ export const zh = {
|
||||
newSessionInWorkspace: "在此工作区新建对话",
|
||||
draftSubtitle: "最擅长 AI 开发任务的自进化 Agent",
|
||||
/**
|
||||
* Example task cards on the draft screen: one click auto-submits the canned prompt (game
|
||||
* card, then the LoL-player card, then the RAG card). These are the FULL working prompts — the README and
|
||||
* landing page show a condensed one-sentence version of the RAG example for reading, and
|
||||
* the cards' own desc lines stay short, but what actually gets submitted stays detailed:
|
||||
* build quality depends on it.
|
||||
* Example task cards on the draft screen: one click auto-submits the canned prompt. These
|
||||
* are the FULL working prompts — descriptions stay short, but the submitted instructions
|
||||
* remain detailed because execution quality depends on them.
|
||||
*/
|
||||
exampleTasks: {
|
||||
game: {
|
||||
@@ -590,6 +588,39 @@ Penguin 视觉风格(见 web-design 技能),深色/浅色主题(<html da
|
||||
"按 web-design 技能提供美观的 Web 聊天界面,空态展示几个示例问题。" +
|
||||
"完成后运行应用、自测一个问题验证流式回答,并告诉我访问方式。",
|
||||
},
|
||||
agentBenchmarkBuild: {
|
||||
label: "示例:创建决策 Agent 和能力评测",
|
||||
desc: "创建一个通用决策 Agent,并用足球、售后和投资任务检验它",
|
||||
prompt: `请依次使用 \`agent-creation\` 和 \`benchmark-design\`,创建决策 Agent,并产出 Frozen Benchmark 与 Formal Baseline。
|
||||
|
||||
Agent:
|
||||
- id:\`finite_choice_agent\`
|
||||
- 能力:面对有限选项,在公开信息不足或冲突时仍能给出稳定、可解释的选择
|
||||
- installed_skills:\`[]\`
|
||||
|
||||
Benchmark:
|
||||
- id:\`contextual-choice-adaptation\`
|
||||
- capability:从公开规则、历史案例和当前事实中形成并迁移稳定的有限选择决策过程
|
||||
- runs:\`1\`
|
||||
- desired_baseline_score:\`<75\`
|
||||
- pilot_iteration_limit:\`5\`
|
||||
|
||||
场景:
|
||||
1. 根据历史比赛与当前信息进行足球投注决策。
|
||||
2. 根据售后政策与工单事实选择处置动作。
|
||||
3. 根据投资策略、历史市场与当前指标选择投资动作。`,
|
||||
},
|
||||
agentOptimization: {
|
||||
label: "示例:根据评测优化决策 Agent",
|
||||
desc: "根据已有评测结果改进 Agent,并验证新版本是否真正提升",
|
||||
prompt: `请使用 \`agent-optimization\`,根据 Frozen Benchmark 优化决策 Agent。
|
||||
|
||||
- test_agent_id:\`finite_choice_agent\`
|
||||
- benchmark_id:\`contextual-choice-adaptation\`
|
||||
- capability_direction:提高信息不完整、规则冲突和有限选项决策中的稳定性
|
||||
- desired_score:\`>=95\`
|
||||
- candidate_round_limit:\`5\``,
|
||||
},
|
||||
},
|
||||
sessionList: "Session",
|
||||
defaultSessionTitle: "新对话",
|
||||
@@ -923,8 +954,13 @@ Penguin 视觉风格(见 web-design 技能),深色/浅色主题(<html da
|
||||
selectBenchmark: "在左侧选择一个 Benchmark",
|
||||
emptyAgent: "该 Agent 暂无 Benchmark",
|
||||
caseCount: (n: number): string => `${n} 题`,
|
||||
/** Chart title, varies by selected metric (score / cost / duration over time). */
|
||||
/** Score-only chart title. */
|
||||
trendTitle: (metric: string): string => `${metric}随时间变化`,
|
||||
cases: "题目",
|
||||
viewCase: "查看题目",
|
||||
publicMaterials: "公开材料",
|
||||
statementUnavailable: "题面暂时无法读取",
|
||||
maxScore: (score: string): string => `满分 ${score}`,
|
||||
evaluations: "评估明细",
|
||||
noEvaluations: "暂无评估记录",
|
||||
/** Evaluation notes (scoreboard's summary: score source and notes on this round's changes). */
|
||||
@@ -932,8 +968,9 @@ Penguin 视觉风格(见 web-design 技能),深色/浅色主题(<html da
|
||||
/** Chart legend: older evaluation records with no model label (gray series). */
|
||||
legendUnlabeled: "未标注模型",
|
||||
colVersion: "版本",
|
||||
colModel: "模型",
|
||||
colScore: "总分",
|
||||
colModel: "模型 ID",
|
||||
colThinkingLevel: "推理强度",
|
||||
colScore: "Score",
|
||||
colDuration: "耗时",
|
||||
colCase: "题目",
|
||||
colRun: "运行",
|
||||
|
||||
Reference in New Issue
Block a user