feat: 完善两阶段 Agent 自进化 Pipeline、Skills 与 Benchmark (#129)
This commit is contained in:
@@ -1214,22 +1214,23 @@ export interface AgentImportResponse {
|
||||
/** Raw result of a single run (a scoreboard per-case runs[] entry). */
|
||||
export interface BenchmarkRunScore {
|
||||
score: number;
|
||||
cost?: number;
|
||||
durationMs?: number;
|
||||
/** Run cost, or null when unavailable. */
|
||||
cost: number | null;
|
||||
durationMs: number;
|
||||
/** Id of the Session under test in this run (links to Trace). */
|
||||
sessionId?: string;
|
||||
sessionId: string;
|
||||
}
|
||||
|
||||
export interface BenchmarkCaseScore {
|
||||
case: string;
|
||||
/** Per-case score = average of runs (equals that single run's score under the legacy single-run format). */
|
||||
/** Model-written average of this Case's Run scores, on the fixed 0..100 scale. */
|
||||
score: number;
|
||||
cost?: number;
|
||||
durationMs?: number;
|
||||
/** For legacy format compatibility: per-case single Session id (new format keeps it inside runs[]). */
|
||||
sessionId?: string;
|
||||
/** Raw results per run; unset under the legacy format (the server backfills one entry when parsing as a single run). */
|
||||
runs?: BenchmarkRunScore[];
|
||||
/** Model-written average of known Run costs; null when every Run cost is unknown. */
|
||||
cost: number | null;
|
||||
/** Model-written average of Run durations, rounded to an integer. */
|
||||
durationMs: number;
|
||||
/** Raw results per Run. */
|
||||
runs: BenchmarkRunScore[];
|
||||
}
|
||||
|
||||
export interface BenchmarkEvaluation {
|
||||
@@ -1240,15 +1241,19 @@ export interface BenchmarkEvaluation {
|
||||
/** Evaluation summary body: how the score was derived, what optimizations were made to the Agent this round (required when generating, tolerated as unset when displaying). */
|
||||
summary?: string;
|
||||
/** Model actually used for this evaluation round (upstream id, paired with provider; the chart series is split by model). */
|
||||
modelId?: string;
|
||||
modelId: string;
|
||||
/** Provider group for `modelId`. */
|
||||
provider?: string;
|
||||
provider: string;
|
||||
/** Thinking level read from the unchanged Target Agent configuration. */
|
||||
thinkingLevel: string;
|
||||
/** Agent State version number under test. */
|
||||
version?: number;
|
||||
/** Total score (sum of per-case scores; max score defined by the scoring rubric). */
|
||||
version: number;
|
||||
/** Model-written average of Case scores, on the fixed 0..100 scale. */
|
||||
score: number;
|
||||
cost?: number;
|
||||
durationMs?: number;
|
||||
/** Model-written average of known Case costs; null when every Case cost is unknown. */
|
||||
cost: number | null;
|
||||
/** Model-written average of Case durations, rounded to an integer. */
|
||||
durationMs: number;
|
||||
cases: BenchmarkCaseScore[];
|
||||
}
|
||||
|
||||
@@ -1270,6 +1275,17 @@ export interface BenchmarksResponse {
|
||||
benchmarks: BenchmarkSummary[];
|
||||
}
|
||||
|
||||
/** Public Benchmark Case metadata. Rubric and Gold content are never included. */
|
||||
export interface BenchmarkCaseSummary {
|
||||
id: string;
|
||||
/** First Markdown heading with an optional leading "Case N:" removed; falls back to id. */
|
||||
title: string;
|
||||
}
|
||||
|
||||
export interface BenchmarkCasesResponse {
|
||||
cases: BenchmarkCaseSummary[];
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Skill library and Agent's installed Skills
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
@@ -154,7 +154,7 @@ export function buildAppDeps(config: ServerConfig, overrides: BuildDepsOverrides
|
||||
// Per-process secret: preview tokens are short-lived, so losing them on restart is
|
||||
// harmless and there is nothing to persist or rotate.
|
||||
const previewTokens = createPreviewTokenSigner();
|
||||
const benchmarks = new BenchmarkService(config.root);
|
||||
const benchmarks = new BenchmarkService(config.root, workspaceFiles);
|
||||
const snapshots = new SnapshotService(config.root);
|
||||
const usageService = new UsageService(
|
||||
usageRepo,
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
/**
|
||||
* Benchmark scoring routes:
|
||||
* GET /api/projects/:p/agents/:a/benchmarks (any member, read-only)
|
||||
* GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases
|
||||
* GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/files
|
||||
* GET /api/projects/:p/agents/:a/benchmarks/:benchmarkId/cases/:caseId/files/content
|
||||
* Returns the Agent's Benchmark list (title/description from benchmark_config.toml)
|
||||
* along with the evaluations[] from scoreboard.yaml.
|
||||
*/
|
||||
@@ -9,6 +12,8 @@ import type { AppEnv } from "../../auth/middleware.js";
|
||||
import type { AppDeps } from "../../app.js";
|
||||
import { requireValidId } from "../validate.js";
|
||||
|
||||
const TEXT_PREVIEW_BYTES = 256 * 1024;
|
||||
|
||||
export function benchmarksRoutes(deps: AppDeps): Hono<AppEnv> {
|
||||
const app = new Hono<AppEnv>();
|
||||
|
||||
@@ -20,5 +25,61 @@ export function benchmarksRoutes(deps: AppDeps): Hono<AppEnv> {
|
||||
return c.json(await deps.benchmarks.list(projectId, agentId));
|
||||
});
|
||||
|
||||
app.get("/:benchmarkId/cases", async (c) => {
|
||||
const projectId = requireValidId(c, "projectId");
|
||||
const agentId = requireValidId(c, "agentId");
|
||||
const benchmarkId = requireValidId(c, "benchmarkId");
|
||||
deps.projectService.requireProjectAccess(c.var.user.userId, projectId);
|
||||
await deps.agentConfigService.requireExists(projectId, agentId);
|
||||
return c.json(await deps.benchmarks.listCases(projectId, agentId, benchmarkId));
|
||||
});
|
||||
|
||||
app.get("/:benchmarkId/cases/:caseId/files", async (c) => {
|
||||
const projectId = requireValidId(c, "projectId");
|
||||
const agentId = requireValidId(c, "agentId");
|
||||
const benchmarkId = requireValidId(c, "benchmarkId");
|
||||
const caseId = requireValidId(c, "caseId");
|
||||
deps.projectService.requireProjectAccess(c.var.user.userId, projectId);
|
||||
await deps.agentConfigService.requireExists(projectId, agentId);
|
||||
return c.json(
|
||||
await deps.benchmarks.listCaseFiles(
|
||||
projectId,
|
||||
agentId,
|
||||
benchmarkId,
|
||||
caseId,
|
||||
c.req.query("path") ?? "",
|
||||
),
|
||||
);
|
||||
});
|
||||
|
||||
app.get("/:benchmarkId/cases/:caseId/files/content", async (c) => {
|
||||
const projectId = requireValidId(c, "projectId");
|
||||
const agentId = requireValidId(c, "agentId");
|
||||
const benchmarkId = requireValidId(c, "benchmarkId");
|
||||
const caseId = requireValidId(c, "caseId");
|
||||
deps.projectService.requireProjectAccess(c.var.user.userId, projectId);
|
||||
await deps.agentConfigService.requireExists(projectId, agentId);
|
||||
const download = c.req.query("download") === "1";
|
||||
const boundedPreview = !download && c.req.query("preview") === "1";
|
||||
const { data, fileName, contentType, scriptable, truncated } =
|
||||
await deps.benchmarks.readCaseFile(
|
||||
projectId,
|
||||
agentId,
|
||||
benchmarkId,
|
||||
caseId,
|
||||
c.req.query("path") ?? "",
|
||||
boundedPreview ? { maxBytes: TEXT_PREVIEW_BYTES } : undefined,
|
||||
);
|
||||
return new Response(new Uint8Array(data), {
|
||||
status: 200,
|
||||
headers: {
|
||||
"Content-Type": !download && scriptable ? "text/plain; charset=utf-8" : contentType,
|
||||
"Content-Disposition": `${download ? "attachment" : "inline"}; filename*=UTF-8''${encodeURIComponent(fileName)}`,
|
||||
"X-Content-Type-Options": "nosniff",
|
||||
...(truncated ? { "X-Content-Truncated": "1" } : {}),
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
return app;
|
||||
}
|
||||
|
||||
@@ -1,16 +1,15 @@
|
||||
/**
|
||||
* Benchmark score reading (read-only display): walks `benchmarks/<id>/`, reads
|
||||
* `benchmark_config.toml` (title,
|
||||
* description, evaluation Model, per-case run count `runs`) and `scoreboard.yaml`
|
||||
* (evaluations[], scoreboard v2: each case carries a runs array and a summary).
|
||||
* `benchmark_config.toml` (title, description, per-case run count `runs`) and
|
||||
* `scoreboard.yaml` (evaluations[], each case carries its model-written averages
|
||||
* and a runs array).
|
||||
* Content is created and refined by benchmark_builder; the server only reads it.
|
||||
* Missing or corrupt files always degrade gracefully (title falls back to the
|
||||
* directory name, scores come back empty) rather than throwing.
|
||||
*
|
||||
* The three per-case metrics trust the file's own values; when missing they're
|
||||
* computed as the average over the runs array. The old format (no runs at the
|
||||
* case level, a single session_id) is parsed as a single run — the server backfills
|
||||
* one run entry.
|
||||
* Case and Evaluation averages are authoritative file values. The server validates
|
||||
* the current shape but never recomputes aggregates and does not migrate or backfill
|
||||
* old Scoreboard formats.
|
||||
* Docs: /docs/self-improvement § "Benchmark storage".
|
||||
*/
|
||||
import fs from "node:fs/promises";
|
||||
@@ -20,11 +19,22 @@ import { parse as parseYaml } from "yaml";
|
||||
import { benchmarksDir } from "@prismshadow/penguin-core";
|
||||
import type {
|
||||
BenchmarkCaseScore,
|
||||
BenchmarkCaseSummary,
|
||||
BenchmarkCasesResponse,
|
||||
BenchmarkEvaluation,
|
||||
BenchmarkRunScore,
|
||||
BenchmarkSummary,
|
||||
BenchmarksResponse,
|
||||
WorkspaceFilesResponse,
|
||||
} from "../api/types.js";
|
||||
import type {
|
||||
WorkspaceFileContent,
|
||||
WorkspaceFileReadOptions,
|
||||
WorkspaceFilesService,
|
||||
} from "./workspace-files-service.js";
|
||||
import { HttpError } from "../http/errors.js";
|
||||
|
||||
const STATEMENT_TITLE_READ_BYTES = 64 * 1024;
|
||||
|
||||
function asRecord(v: unknown): Record<string, unknown> {
|
||||
return v !== null && typeof v === "object" && !Array.isArray(v)
|
||||
@@ -36,110 +46,151 @@ function numberOr(v: unknown): number | undefined {
|
||||
return typeof v === "number" && Number.isFinite(v) ? v : undefined;
|
||||
}
|
||||
|
||||
function scoreOr(v: unknown): number | undefined {
|
||||
const value = numberOr(v);
|
||||
return value !== undefined && value >= 0 && value <= 100 ? value : undefined;
|
||||
}
|
||||
|
||||
function nonNegativeOr(v: unknown): number | undefined {
|
||||
const value = numberOr(v);
|
||||
return value !== undefined && value >= 0 ? value : undefined;
|
||||
}
|
||||
|
||||
function nonNegativeIntegerOr(v: unknown): number | undefined {
|
||||
const value = nonNegativeOr(v);
|
||||
return value !== undefined && Number.isInteger(value) ? value : undefined;
|
||||
}
|
||||
|
||||
/** `null` is the one valid unknown-cost representation; undefined means invalid input. */
|
||||
function nullableCostOr(v: unknown): number | null | undefined {
|
||||
if (v === null) return null;
|
||||
return nonNegativeOr(v);
|
||||
}
|
||||
|
||||
function stringOr(v: unknown): string | undefined {
|
||||
return typeof v === "string" && v !== "" ? v : undefined;
|
||||
}
|
||||
|
||||
/** Shapes a single run entry: score is the minimum requirement, other fields tolerate being absent; a bad entry returns null and is dropped. */
|
||||
function toRun(v: unknown): BenchmarkRunScore | null {
|
||||
const r = asRecord(v);
|
||||
const score = numberOr(r.score);
|
||||
if (score === undefined) return null;
|
||||
const cost = numberOr(r.cost);
|
||||
const durationMs = numberOr(r.duration_ms);
|
||||
const sessionId = stringOr(r.session_id);
|
||||
return {
|
||||
score,
|
||||
...(cost !== undefined ? { cost } : {}),
|
||||
...(durationMs !== undefined ? { durationMs } : {}),
|
||||
...(sessionId !== undefined ? { sessionId } : {}),
|
||||
};
|
||||
function isWithin(parent: string, child: string): boolean {
|
||||
const relative = path.relative(parent, child);
|
||||
return relative !== "" && !relative.startsWith("..") && !path.isAbsolute(relative);
|
||||
}
|
||||
|
||||
/** Average of a metric across runs; undefined when there's no value at all (never forced to 0). */
|
||||
function averageOf(runs: BenchmarkRunScore[], pick: (r: BenchmarkRunScore) => number | undefined) {
|
||||
const values = runs.map(pick).filter((v): v is number => v !== undefined);
|
||||
if (values.length === 0) return undefined;
|
||||
return values.reduce((a, b) => a + b, 0) / values.length;
|
||||
function statementTitle(statement: string, fallback: string): string {
|
||||
const heading = /^#\s+(.+)$/m.exec(statement)?.[1]?.trim();
|
||||
return heading?.replace(/^Case\s+\d+\s*:\s*/i, "") || fallback;
|
||||
}
|
||||
|
||||
async function readStatementTitle(readme: string, fallback: string): Promise<string> {
|
||||
const handle = await fs.open(readme, "r");
|
||||
try {
|
||||
const buffer = Buffer.alloc(STATEMENT_TITLE_READ_BYTES);
|
||||
const { bytesRead } = await handle.read(buffer, 0, buffer.length, 0);
|
||||
return statementTitle(buffer.subarray(0, bytesRead).toString("utf8"), fallback);
|
||||
} finally {
|
||||
await handle.close();
|
||||
}
|
||||
}
|
||||
|
||||
/** Shapes one current-format Run; a malformed entry invalidates its containing Case. */
|
||||
function toRun(v: unknown): BenchmarkRunScore | null {
|
||||
const r = asRecord(v);
|
||||
const score = scoreOr(r.score);
|
||||
const cost = nullableCostOr(r.cost);
|
||||
const durationMs = nonNegativeIntegerOr(r.duration_ms);
|
||||
const sessionId = stringOr(r.session_id);
|
||||
if (score === undefined || cost === undefined || durationMs === undefined || !sessionId)
|
||||
return null;
|
||||
return { score, cost, durationMs, sessionId };
|
||||
}
|
||||
|
||||
/**
|
||||
* Shapes a case-level entry (scoreboard v2): the three metrics trust the file's own
|
||||
* values, falling back to an average over runs when missing; the old format (no
|
||||
* runs, a single case-level session_id) is backfilled into a single run. case and a
|
||||
* score (from the file or derivable from runs) are the minimum requirement,
|
||||
* otherwise the entry is dropped.
|
||||
* Shapes one current-format Case. Its stored aggregates are trusted as written:
|
||||
* this parser intentionally performs no average or consistency calculation.
|
||||
*/
|
||||
function toCase(v: unknown): BenchmarkCaseScore | null {
|
||||
const cr = asRecord(v);
|
||||
const caseId = stringOr(cr.case);
|
||||
if (caseId === undefined) return null;
|
||||
const parsedRuns = Array.isArray(cr.runs)
|
||||
? cr.runs.map(toRun).filter((r): r is BenchmarkRunScore => r !== null)
|
||||
: [];
|
||||
const score = numberOr(cr.score) ?? averageOf(parsedRuns, (r) => r.score);
|
||||
if (score === undefined) return null;
|
||||
const cost = numberOr(cr.cost) ?? averageOf(parsedRuns, (r) => r.cost);
|
||||
const durationMs = numberOr(cr.duration_ms) ?? averageOf(parsedRuns, (r) => r.durationMs);
|
||||
const sessionId = stringOr(cr.session_id);
|
||||
const runs: BenchmarkRunScore[] =
|
||||
parsedRuns.length > 0
|
||||
? parsedRuns
|
||||
: [
|
||||
// The old format is parsed as a single run: the case-level values are that run's raw result.
|
||||
{
|
||||
score,
|
||||
...(cost !== undefined ? { cost } : {}),
|
||||
...(durationMs !== undefined ? { durationMs } : {}),
|
||||
...(sessionId !== undefined ? { sessionId } : {}),
|
||||
},
|
||||
];
|
||||
const score = scoreOr(cr.score);
|
||||
const cost = nullableCostOr(cr.cost);
|
||||
const durationMs = nonNegativeIntegerOr(cr.duration_ms);
|
||||
if (
|
||||
!caseId ||
|
||||
score === undefined ||
|
||||
cost === undefined ||
|
||||
durationMs === undefined ||
|
||||
"max_score" in cr ||
|
||||
!Array.isArray(cr.runs) ||
|
||||
cr.runs.length === 0
|
||||
) {
|
||||
return null;
|
||||
}
|
||||
const parsedRuns = cr.runs.map(toRun);
|
||||
if (parsedRuns.some((run) => run === null)) return null;
|
||||
const runs = parsedRuns as BenchmarkRunScore[];
|
||||
return {
|
||||
case: caseId,
|
||||
score,
|
||||
...(cost !== undefined ? { cost } : {}),
|
||||
...(durationMs !== undefined ? { durationMs } : {}),
|
||||
...(sessionId !== undefined ? { sessionId } : {}),
|
||||
cost,
|
||||
durationMs,
|
||||
runs,
|
||||
};
|
||||
}
|
||||
|
||||
/** Shapes a single evaluation record: time and score are the minimum requirement, other fields (summary, etc.) tolerate being absent. */
|
||||
/** Shapes one current-format Evaluation and trusts its stored aggregate metrics. */
|
||||
function toEvaluation(v: unknown): BenchmarkEvaluation | null {
|
||||
const r = asRecord(v);
|
||||
const time = r.time instanceof Date ? r.time.toISOString() : r.time;
|
||||
const score = numberOr(r.score);
|
||||
if (typeof time !== "string" || time === "" || score === undefined) return null;
|
||||
const cases: BenchmarkCaseScore[] = Array.isArray(r.cases)
|
||||
? r.cases.map(toCase).filter((c): c is BenchmarkCaseScore => c !== null)
|
||||
: [];
|
||||
const score = scoreOr(r.score);
|
||||
const cost = nullableCostOr(r.cost);
|
||||
const durationMs = nonNegativeIntegerOr(r.duration_ms);
|
||||
const summary = stringOr(r.summary);
|
||||
// Title and body are separate: summary_title is a one-line
|
||||
// conclusion, summary is the body text.
|
||||
const summaryTitle = stringOr(r.summary_title);
|
||||
// The Model actually used for this evaluation run (paired with provider):
|
||||
// charted curves are split into series by model, each with a distinct color.
|
||||
const modelId = stringOr(r.model_id);
|
||||
const provider = stringOr(r.provider);
|
||||
const version = numberOr(r.version);
|
||||
const cost = numberOr(r.cost);
|
||||
const durationMs = numberOr(r.duration_ms);
|
||||
const thinkingLevel = stringOr(r.thinking_level);
|
||||
const version = nonNegativeIntegerOr(r.version);
|
||||
if (
|
||||
typeof time !== "string" ||
|
||||
time === "" ||
|
||||
score === undefined ||
|
||||
cost === undefined ||
|
||||
durationMs === undefined ||
|
||||
!modelId ||
|
||||
!provider ||
|
||||
!thinkingLevel ||
|
||||
version === undefined ||
|
||||
version < 1 ||
|
||||
!Array.isArray(r.cases) ||
|
||||
r.cases.length === 0
|
||||
) {
|
||||
return null;
|
||||
}
|
||||
const parsedCases = r.cases.map(toCase);
|
||||
if (parsedCases.some((item) => item === null)) return null;
|
||||
const cases = parsedCases as BenchmarkCaseScore[];
|
||||
return {
|
||||
time,
|
||||
...(summaryTitle !== undefined ? { summaryTitle } : {}),
|
||||
...(summary !== undefined ? { summary } : {}),
|
||||
...(modelId !== undefined ? { modelId } : {}),
|
||||
...(provider !== undefined ? { provider } : {}),
|
||||
modelId,
|
||||
provider,
|
||||
thinkingLevel,
|
||||
score,
|
||||
...(version !== undefined ? { version } : {}),
|
||||
...(cost !== undefined ? { cost } : {}),
|
||||
...(durationMs !== undefined ? { durationMs } : {}),
|
||||
version,
|
||||
cost,
|
||||
durationMs,
|
||||
cases,
|
||||
};
|
||||
}
|
||||
|
||||
export class BenchmarkService {
|
||||
constructor(private readonly root: string) {}
|
||||
constructor(
|
||||
private readonly root: string,
|
||||
private readonly workspaceFiles: WorkspaceFilesService,
|
||||
) {}
|
||||
|
||||
async list(projectId: string, agentId: string): Promise<BenchmarksResponse> {
|
||||
const dir = benchmarksDir(this.root, projectId, agentId);
|
||||
@@ -157,6 +208,98 @@ export class BenchmarkService {
|
||||
return { benchmarks };
|
||||
}
|
||||
|
||||
async listCases(
|
||||
projectId: string,
|
||||
agentId: string,
|
||||
benchmarkId: string,
|
||||
): Promise<BenchmarkCasesResponse> {
|
||||
const baseDir = benchmarksDir(this.root, projectId, agentId);
|
||||
const benchDir = path.join(baseDir, benchmarkId);
|
||||
let entries: Array<{ name: string; isDirectory(): boolean }>;
|
||||
let realBaseDir: string;
|
||||
let realBenchDir: string;
|
||||
try {
|
||||
[entries, realBaseDir, realBenchDir] = await Promise.all([
|
||||
fs.readdir(benchDir, { withFileTypes: true }),
|
||||
fs.realpath(baseDir),
|
||||
fs.realpath(benchDir),
|
||||
]);
|
||||
} catch {
|
||||
return { cases: [] };
|
||||
}
|
||||
if (!isWithin(realBaseDir, realBenchDir)) return { cases: [] };
|
||||
|
||||
const cases: BenchmarkCaseSummary[] = [];
|
||||
for (const entry of entries
|
||||
.filter((item) => item.isDirectory() && item.name.startsWith("CASE-"))
|
||||
.sort((a, b) => a.name.localeCompare(b.name))) {
|
||||
const fallback: BenchmarkCaseSummary = { id: entry.name, title: entry.name };
|
||||
try {
|
||||
const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, entry.name);
|
||||
const realReadme = await fs.realpath(path.join(statementDir, "README.md"));
|
||||
if (!isWithin(statementDir, realReadme)) throw new Error("README escapes Statement");
|
||||
cases.push({
|
||||
id: entry.name,
|
||||
title: await readStatementTitle(realReadme, entry.name),
|
||||
});
|
||||
} catch {
|
||||
cases.push(fallback);
|
||||
}
|
||||
}
|
||||
return { cases };
|
||||
}
|
||||
|
||||
async listCaseFiles(
|
||||
projectId: string,
|
||||
agentId: string,
|
||||
benchmarkId: string,
|
||||
caseId: string,
|
||||
rel: string,
|
||||
): Promise<WorkspaceFilesResponse> {
|
||||
const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, caseId);
|
||||
return this.workspaceFiles.list(statementDir, rel);
|
||||
}
|
||||
|
||||
async readCaseFile(
|
||||
projectId: string,
|
||||
agentId: string,
|
||||
benchmarkId: string,
|
||||
caseId: string,
|
||||
rel: string,
|
||||
options?: WorkspaceFileReadOptions,
|
||||
): Promise<WorkspaceFileContent> {
|
||||
const statementDir = await this.statementRoot(projectId, agentId, benchmarkId, caseId);
|
||||
return this.workspaceFiles.read(statementDir, rel, options);
|
||||
}
|
||||
|
||||
private async statementRoot(
|
||||
projectId: string,
|
||||
agentId: string,
|
||||
benchmarkId: string,
|
||||
caseId: string,
|
||||
): Promise<string> {
|
||||
const benchDir = path.join(benchmarksDir(this.root, projectId, agentId), benchmarkId);
|
||||
const caseDir = path.join(benchDir, caseId);
|
||||
const statementDir = path.join(caseDir, "statement");
|
||||
try {
|
||||
const [realBenchDir, realCaseDir, realStatementDir] = await Promise.all([
|
||||
fs.realpath(benchDir),
|
||||
fs.realpath(caseDir),
|
||||
fs.realpath(statementDir),
|
||||
]);
|
||||
if (
|
||||
!isWithin(realBenchDir, realCaseDir) ||
|
||||
path.dirname(realStatementDir) !== realCaseDir ||
|
||||
path.basename(realStatementDir) !== "statement"
|
||||
) {
|
||||
throw new Error("Statement path is not canonical");
|
||||
}
|
||||
return realStatementDir;
|
||||
} catch {
|
||||
throw new HttpError(404, "not_found", "Public Case Statement does not exist.");
|
||||
}
|
||||
}
|
||||
|
||||
private async readBenchmark(benchDir: string, id: string): Promise<BenchmarkSummary> {
|
||||
// benchmark_config.toml: title, description, and per-case run count (falls back
|
||||
// to defaults if corrupt). The model isn't part of the config — each evaluation
|
||||
@@ -195,11 +338,11 @@ export class BenchmarkService {
|
||||
// No scores yet.
|
||||
}
|
||||
|
||||
// Case count: number of case subfolders (the statement/rubric structure isn't validated here).
|
||||
// Case count: number of semantic Case subfolders.
|
||||
let caseCount = 0;
|
||||
try {
|
||||
const entries = await fs.readdir(benchDir, { withFileTypes: true });
|
||||
caseCount = entries.filter((e) => e.isDirectory()).length;
|
||||
caseCount = entries.filter((e) => e.isDirectory() && e.name.startsWith("CASE-")).length;
|
||||
} catch {
|
||||
// Stays at 0.
|
||||
}
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
import fs from "node:fs/promises";
|
||||
import { constants as fsc } from "node:fs";
|
||||
import path from "node:path";
|
||||
import type { WorkspaceFilesResponse } from "../api/types.js";
|
||||
import type { WorkspaceFileEntry, WorkspaceFilesResponse } from "../api/types.js";
|
||||
import { HttpError } from "../http/errors.js";
|
||||
import { badRequest } from "../http/validate.js";
|
||||
|
||||
@@ -48,6 +48,13 @@ export interface WorkspaceFileContent {
|
||||
contentType: string;
|
||||
/** Types whose same-origin inline rendering would execute scripts (html/svg): inline preview must fall back to plain text. */
|
||||
scriptable: boolean;
|
||||
/** True when a bounded preview returned only the beginning of the file. */
|
||||
truncated?: boolean;
|
||||
}
|
||||
|
||||
export interface WorkspaceFileReadOptions {
|
||||
/** Return at most this many bytes. Used for bounded text previews. */
|
||||
maxBytes?: number;
|
||||
}
|
||||
|
||||
export class WorkspaceFilesService {
|
||||
@@ -181,9 +188,14 @@ export class WorkspaceFilesService {
|
||||
return unique.filter((_, i) => exists[i]);
|
||||
}
|
||||
|
||||
/** List a directory: dirs come first, each group sorted by name; kind follows the symlink target (consistent with read behavior). */
|
||||
/**
|
||||
* List a directory: dirs come first, each group sorted by name. Entries whose
|
||||
* canonical target leaves the Workspace are omitted, so listing cannot expose
|
||||
* metadata for an out-of-bounds symlink.
|
||||
*/
|
||||
async list(workspace: string, rel: string): Promise<WorkspaceFilesResponse> {
|
||||
const dir = await this.resolveRead(workspace, rel);
|
||||
const realBase = await this.realBase(workspace);
|
||||
let dirents;
|
||||
try {
|
||||
dirents = await fs.readdir(dir, { withFileTypes: true });
|
||||
@@ -196,28 +208,24 @@ export class WorkspaceFilesService {
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
const entries = await Promise.all(
|
||||
dirents.map(async (d) => {
|
||||
let sizeBytes = 0;
|
||||
let mtime = "";
|
||||
// Dirent doesn't report the target type for a symlink, so stat (following the link) is used to determine dir/file.
|
||||
let isDir = d.isDirectory();
|
||||
const listed = await Promise.all(
|
||||
dirents.map(async (d): Promise<WorkspaceFileEntry | null> => {
|
||||
try {
|
||||
const stat = await fs.stat(path.join(dir, d.name));
|
||||
sizeBytes = stat.size;
|
||||
mtime = stat.mtime.toISOString();
|
||||
isDir = stat.isDirectory();
|
||||
const canonical = await fs.realpath(path.join(dir, d.name));
|
||||
this.assertInside(canonical, realBase);
|
||||
const stat = await fs.stat(canonical);
|
||||
return {
|
||||
name: d.name,
|
||||
kind: stat.isDirectory() ? "dir" : "file",
|
||||
sizeBytes: stat.size,
|
||||
mtime: stat.mtime.toISOString(),
|
||||
};
|
||||
} catch {
|
||||
// A dangling symlink or similar: keep the entry, with size/time left at defaults.
|
||||
return null;
|
||||
}
|
||||
return {
|
||||
name: d.name,
|
||||
kind: isDir ? ("dir" as const) : ("file" as const),
|
||||
sizeBytes,
|
||||
mtime,
|
||||
};
|
||||
}),
|
||||
);
|
||||
const entries = listed.filter((entry): entry is WorkspaceFileEntry => entry !== null);
|
||||
entries.sort((a, b) =>
|
||||
a.kind === b.kind ? a.name.localeCompare(b.name) : a.kind === "dir" ? -1 : 1,
|
||||
);
|
||||
@@ -225,7 +233,11 @@ export class WorkspaceFilesService {
|
||||
}
|
||||
|
||||
/** Read a file (preview/download): IO on the canonical path (resolveRead has already eliminated symlink escapes). */
|
||||
async read(workspace: string, rel: string): Promise<WorkspaceFileContent> {
|
||||
async read(
|
||||
workspace: string,
|
||||
rel: string,
|
||||
options?: WorkspaceFileReadOptions,
|
||||
): Promise<WorkspaceFileContent> {
|
||||
const file = await this.resolveRead(workspace, rel);
|
||||
let stat;
|
||||
try {
|
||||
@@ -234,16 +246,38 @@ export class WorkspaceFilesService {
|
||||
throw new HttpError(404, "path_not_found", "File does not exist.");
|
||||
}
|
||||
if (stat.isDirectory()) throw badRequest("path is a directory.");
|
||||
if (stat.size > MAX_READ_BYTES) {
|
||||
const maxBytes = options?.maxBytes;
|
||||
if (
|
||||
maxBytes !== undefined &&
|
||||
(!Number.isSafeInteger(maxBytes) || maxBytes < 1 || maxBytes > MAX_READ_BYTES)
|
||||
) {
|
||||
throw badRequest("maxBytes must be a positive integer within the read limit.");
|
||||
}
|
||||
if (maxBytes === undefined && stat.size > MAX_READ_BYTES) {
|
||||
throw new HttpError(413, "file_too_large", "File exceeds the 50MB read limit.");
|
||||
}
|
||||
const data = await fs.readFile(file);
|
||||
let data: Buffer;
|
||||
let truncated = false;
|
||||
if (maxBytes !== undefined && stat.size > maxBytes) {
|
||||
const handle = await fs.open(file, "r");
|
||||
try {
|
||||
const buffer = Buffer.alloc(maxBytes);
|
||||
const { bytesRead } = await handle.read(buffer, 0, maxBytes, 0);
|
||||
data = buffer.subarray(0, bytesRead);
|
||||
truncated = true;
|
||||
} finally {
|
||||
await handle.close();
|
||||
}
|
||||
} else {
|
||||
data = await fs.readFile(file);
|
||||
}
|
||||
const ext = path.extname(file).toLowerCase();
|
||||
return {
|
||||
data,
|
||||
fileName: path.basename(file),
|
||||
contentType: CONTENT_TYPES[ext] ?? "application/octet-stream",
|
||||
scriptable: ext === ".html" || ext === ".htm" || ext === ".svg",
|
||||
...(truncated ? { truncated: true } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -1,10 +1,9 @@
|
||||
/**
|
||||
* Benchmark scoreboard read integration tests (read-only display): benchmark_config.toml title/description and runs
|
||||
* pass-through (falls back to directory name if missing), scoreboard.yaml v2's
|
||||
* evaluations[] (summary pass-through, per-case runs array; per-case metrics trust the
|
||||
* file values, falling back to an average over runs when missing), the legacy format
|
||||
* (per-case single session_id) parsed as a single run and backfilled, bad entries
|
||||
* discarded, case count, empty when unconfigured, permissions (members can read,
|
||||
* evaluations[] (summary pass-through, model-written Case/Evaluation averages and per-case
|
||||
* runs arrays), rejection of legacy Scoreboard entries, case count, empty when
|
||||
* unconfigured, permissions (members can read,
|
||||
* outsiders get 404).
|
||||
*
|
||||
* Tested with a plain Agent (no sample Benchmark pre-installed); default_agent's sample
|
||||
@@ -14,7 +13,12 @@ import fs from "node:fs/promises";
|
||||
import path from "node:path";
|
||||
import { afterEach, beforeEach, describe, expect, it } from "vitest";
|
||||
import { benchmarksDir } from "@prismshadow/penguin-core";
|
||||
import type { BenchmarksResponse, ProjectCreateResponse } from "../src/api/types.js";
|
||||
import type {
|
||||
BenchmarkCasesResponse,
|
||||
BenchmarksResponse,
|
||||
ProjectCreateResponse,
|
||||
WorkspaceFilesResponse,
|
||||
} from "../src/api/types.js";
|
||||
import { apiClient, createTestApp, provisionUser } from "./helpers.js";
|
||||
import type { TestApp } from "./helpers.js";
|
||||
|
||||
@@ -59,10 +63,48 @@ describe("benchmarks api", () => {
|
||||
});
|
||||
});
|
||||
|
||||
it("scoreboard v2: summary/runs pass through; metrics trust file or average runs", async () => {
|
||||
it("current scoreboard: model-written averages, runtime, and runs pass through", async () => {
|
||||
const dir = path.join(benchmarksDir(t.root, projectId, AGENT), "swe-bench-v2");
|
||||
await fs.mkdir(path.join(dir, "CASE-001-excel-task", "statement"), { recursive: true });
|
||||
await fs.mkdir(path.join(dir, "CASE-001-excel-task", "statement", "assets"), {
|
||||
recursive: true,
|
||||
});
|
||||
await fs.mkdir(path.join(dir, "CASE-002-web-task", "statement"), { recursive: true });
|
||||
await fs.mkdir(path.join(dir, "CASE-002-web-task", "rubric"), { recursive: true });
|
||||
await fs.writeFile(
|
||||
path.join(dir, "CASE-001-excel-task", "statement", "README.md"),
|
||||
"# Case 001: Excel cleanup\n\nClean the workbook.",
|
||||
"utf8",
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(dir, "CASE-001-excel-task", "statement", "data.csv"),
|
||||
"id,value\n1,alpha\n",
|
||||
"utf8",
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(dir, "CASE-001-excel-task", "statement", "assets", "notes.txt"),
|
||||
"public notes",
|
||||
"utf8",
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(dir, "CASE-001-excel-task", "statement", "large.txt"),
|
||||
"x".repeat(300 * 1024),
|
||||
"utf8",
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(dir, "CASE-002-web-task", "statement", "README.md"),
|
||||
"# Web task\n\nBuild the page.",
|
||||
"utf8",
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(dir, "CASE-002-web-task", "rubric", "README.md"),
|
||||
"PRIVATE GOLD: never return this text",
|
||||
"utf8",
|
||||
);
|
||||
await fs.symlink(
|
||||
path.join(dir, "CASE-002-web-task", "rubric", "README.md"),
|
||||
path.join(dir, "CASE-001-excel-task", "statement", "private-link.md"),
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(dir, "benchmark_config.toml"),
|
||||
`title = "SWE Bench v2"\ndescription = "Example"\nruns = 2\n`,
|
||||
@@ -76,35 +118,39 @@ describe("benchmarks api", () => {
|
||||
" version: 3",
|
||||
' provider: "deepseek"',
|
||||
' model_id: "deepseek-v4-pro"',
|
||||
' thinking_level: "medium"',
|
||||
' summary_title: "Added planning steps to the system Prompt"',
|
||||
' summary: "Each case run twice and averaged; added planning steps."',
|
||||
" score: 8.0",
|
||||
" cost: 0.05",
|
||||
" duration_ms: 120000",
|
||||
" score: 72.35",
|
||||
" cost: 0.04",
|
||||
" duration_ms: 42500",
|
||||
" cases:",
|
||||
// Per-case metrics are all present: trust the file values (no recomputation even if inconsistent with the runs average).
|
||||
// Stored averages are authoritative even when inconsistent with the raw Runs.
|
||||
' - case: "CASE-001-excel-task"',
|
||||
" score: 4.2",
|
||||
" cost: 0.02",
|
||||
" score: 80.2",
|
||||
" cost: 0.04",
|
||||
" duration_ms: 50000",
|
||||
" runs:",
|
||||
" - score: 4.0",
|
||||
" cost: 0.018",
|
||||
" - score: 80",
|
||||
" cost: null",
|
||||
" duration_ms: 48000",
|
||||
' session_id: "session-run-1"',
|
||||
" - score: 4.5",
|
||||
" cost: 0.022",
|
||||
" - score: 82",
|
||||
" cost: 0.04",
|
||||
" duration_ms: 52000",
|
||||
' session_id: "session-run-2"',
|
||||
// Per-case metrics are missing: computed as the average over runs.
|
||||
// All unknown Run costs produce a model-written null Case cost; the Evaluation ignores it.
|
||||
' - case: "CASE-002-web-task"',
|
||||
" score: 64.5",
|
||||
" cost: null",
|
||||
" duration_ms: 35000",
|
||||
" runs:",
|
||||
" - score: 3.0",
|
||||
" cost: 0.01",
|
||||
" - score: 60",
|
||||
" cost: null",
|
||||
" duration_ms: 30000",
|
||||
' session_id: "session-run-3"',
|
||||
" - score: 4.0",
|
||||
" cost: 0.03",
|
||||
" - score: 70",
|
||||
" cost: null",
|
||||
" duration_ms: 40000",
|
||||
' session_id: "session-run-4"',
|
||||
].join("\n"),
|
||||
@@ -127,28 +173,100 @@ describe("benchmarks api", () => {
|
||||
// The evaluation entry carries this run's model (as a pair) and a summary title (curve series / title-body are displayed separately).
|
||||
expect(evaluation.provider).toBe("deepseek");
|
||||
expect(evaluation.modelId).toBe("deepseek-v4-pro");
|
||||
expect(evaluation.thinkingLevel).toBe("medium");
|
||||
expect(evaluation.summaryTitle).toBe("Added planning steps to the system Prompt");
|
||||
expect(evaluation.summary).toBe("Each case run twice and averaged; added planning steps.");
|
||||
expect(evaluation.score).toBe(8.0);
|
||||
// Per-case metrics are all present: trust the file (4.2, not the runs average of 4.25).
|
||||
expect(evaluation.score).toBe(72.35);
|
||||
expect(evaluation.cost).toBe(0.04);
|
||||
expect(evaluation.durationMs).toBe(42500);
|
||||
expect("maxScore" in evaluation).toBe(false);
|
||||
// Per-case metrics trust the file (80.2, not the Runs' arithmetic mean of 81).
|
||||
const full = evaluation.cases.find((c) => c.case === "CASE-001-excel-task")!;
|
||||
expect(full.score).toBe(4.2);
|
||||
expect(full.cost).toBe(0.02);
|
||||
expect(full.score).toBe(80.2);
|
||||
expect(full.cost).toBe(0.04);
|
||||
expect(full.durationMs).toBe(50000);
|
||||
expect(full.runs).toEqual([
|
||||
{ score: 4.0, cost: 0.018, durationMs: 48000, sessionId: "session-run-1" },
|
||||
{ score: 4.5, cost: 0.022, durationMs: 52000, sessionId: "session-run-2" },
|
||||
{ score: 80, cost: null, durationMs: 48000, sessionId: "session-run-1" },
|
||||
{ score: 82, cost: 0.04, durationMs: 52000, sessionId: "session-run-2" },
|
||||
]);
|
||||
// Per-case metrics are missing: computed as the average over runs.
|
||||
const derived = evaluation.cases.find((c) => c.case === "CASE-002-web-task")!;
|
||||
expect(derived.score).toBe(3.5);
|
||||
expect(derived.cost).toBeCloseTo(0.02, 10);
|
||||
expect(derived.durationMs).toBe(35000);
|
||||
expect(derived.runs).toHaveLength(2);
|
||||
expect(derived.sessionId).toBeUndefined();
|
||||
const partialCost = evaluation.cases.find((c) => c.case === "CASE-002-web-task")!;
|
||||
expect(partialCost.score).toBe(64.5);
|
||||
expect(partialCost.cost).toBeNull();
|
||||
expect(partialCost.durationMs).toBe(35000);
|
||||
expect(partialCost.runs).toEqual([
|
||||
{ score: 60, cost: null, durationMs: 30000, sessionId: "session-run-3" },
|
||||
{ score: 70, cost: null, durationMs: 40000, sessionId: "session-run-4" },
|
||||
]);
|
||||
|
||||
const caseResponse = (await (
|
||||
await member.get(`${base}/swe-bench-v2/cases`)
|
||||
).json()) as BenchmarkCasesResponse;
|
||||
expect(caseResponse).toEqual({
|
||||
cases: [
|
||||
{
|
||||
id: "CASE-001-excel-task",
|
||||
title: "Excel cleanup",
|
||||
},
|
||||
{
|
||||
id: "CASE-002-web-task",
|
||||
title: "Web task",
|
||||
},
|
||||
],
|
||||
});
|
||||
expect(JSON.stringify(caseResponse)).not.toContain("PRIVATE GOLD");
|
||||
|
||||
const filesBase = `${base}/swe-bench-v2/cases/CASE-001-excel-task/files`;
|
||||
const files = (await (await member.get(filesBase)).json()) as WorkspaceFilesResponse;
|
||||
expect(files.path).toBe("");
|
||||
expect(files.entries.map((entry) => `${entry.kind}:${entry.name}`)).toEqual([
|
||||
"dir:assets",
|
||||
"file:data.csv",
|
||||
"file:large.txt",
|
||||
"file:README.md",
|
||||
]);
|
||||
expect(JSON.stringify(files)).not.toContain("private-link.md");
|
||||
expect(
|
||||
(
|
||||
await member.get(
|
||||
`${filesBase}/content?path=${encodeURIComponent("private-link.md")}&preview=1`,
|
||||
)
|
||||
).status,
|
||||
).toBe(400);
|
||||
|
||||
const nested = (await (
|
||||
await member.get(`${filesBase}?path=${encodeURIComponent("assets")}`)
|
||||
).json()) as WorkspaceFilesResponse;
|
||||
expect(nested.entries.map((entry) => entry.name)).toEqual(["notes.txt"]);
|
||||
|
||||
const statement = await member.get(
|
||||
`${filesBase}/content?path=${encodeURIComponent("README.md")}&preview=1`,
|
||||
);
|
||||
expect(statement.status).toBe(200);
|
||||
expect(statement.headers.get("content-type")).toContain("markdown");
|
||||
expect(await statement.text()).toBe("# Case 001: Excel cleanup\n\nClean the workbook.");
|
||||
|
||||
const large = await member.get(
|
||||
`${filesBase}/content?path=${encodeURIComponent("large.txt")}&preview=1`,
|
||||
);
|
||||
expect(large.status).toBe(200);
|
||||
expect(large.headers.get("x-content-truncated")).toBe("1");
|
||||
expect((await large.text()).length).toBe(256 * 1024);
|
||||
|
||||
const download = await member.get(
|
||||
`${filesBase}/content?path=${encodeURIComponent("data.csv")}&download=1`,
|
||||
);
|
||||
expect(download.headers.get("content-disposition")).toContain("attachment");
|
||||
expect(await download.text()).toBe("id,value\n1,alpha\n");
|
||||
|
||||
expect(
|
||||
(await member.get(`${filesBase}/content?path=${encodeURIComponent("../rubric/README.md")}`))
|
||||
.status,
|
||||
).toBe(400);
|
||||
expect((await outsider.get(`${base}/swe-bench-v2/cases`)).status).toBe(404);
|
||||
expect((await outsider.get(filesBase)).status).toBe(404);
|
||||
});
|
||||
|
||||
it("legacy per-case session_id parsed as one backfilled run; bad entries dropped", async () => {
|
||||
it("does not migrate or backfill legacy Scoreboard entries", async () => {
|
||||
const dir = path.join(benchmarksDir(t.root, projectId, AGENT), "swe-bench-v1");
|
||||
await fs.mkdir(path.join(dir, "CASE-001-excel-task", "statement"), { recursive: true });
|
||||
await fs.writeFile(path.join(dir, "benchmark_config.toml"), `title = "SWE Bench v1"\n`, "utf8");
|
||||
@@ -184,26 +302,7 @@ describe("benchmarks api", () => {
|
||||
const bench = res.benchmarks[1]!;
|
||||
expect(bench).toMatchObject({ title: "SWE Bench v1", caseCount: 1 });
|
||||
expect("runs" in bench).toBe(false);
|
||||
expect(bench.evaluations).toHaveLength(1);
|
||||
expect(bench.evaluations[0]).toMatchObject({
|
||||
time: "2026-07-16T10:00:00Z",
|
||||
version: 1,
|
||||
score: 62.5,
|
||||
cost: 1.25,
|
||||
durationMs: 60000,
|
||||
});
|
||||
expect("summary" in bench.evaluations[0]!).toBe(false);
|
||||
// Legacy per-case format: fields unchanged, plus one backfilled run matching the case-level values (the frontend uniformly expands via runs).
|
||||
expect(bench.evaluations[0]?.cases).toEqual([
|
||||
{
|
||||
case: "CASE-001-excel-task",
|
||||
score: 30,
|
||||
cost: 0.5,
|
||||
durationMs: 20000,
|
||||
sessionId: "session-abc",
|
||||
runs: [{ score: 30, cost: 0.5, durationMs: 20000, sessionId: "session-abc" }],
|
||||
},
|
||||
]);
|
||||
expect(bench.evaluations).toEqual([]);
|
||||
expect(res.benchmarks[0]).toMatchObject({
|
||||
title: "empty-bench",
|
||||
caseCount: 0,
|
||||
|
||||
@@ -88,6 +88,7 @@ describe("built-in Agent provisioning", () => {
|
||||
expect(bench.caseCount).toBe(2);
|
||||
expect(bench.evaluations).toHaveLength(3);
|
||||
for (const evaluation of bench.evaluations) {
|
||||
expect(evaluation.thinkingLevel).toBe("medium");
|
||||
expect(evaluation.summary).toBeTruthy();
|
||||
expect(evaluation.cases).toHaveLength(2);
|
||||
for (const c of evaluation.cases) {
|
||||
|
||||
@@ -45,6 +45,9 @@ describe("workspace-files-service", () => {
|
||||
const file = await svc.read(ws, "sub/c.md");
|
||||
expect(file.data.toString()).toBe("# md");
|
||||
expect(file.contentType).toContain("markdown");
|
||||
const preview = await svc.read(ws, "sub/c.md", { maxBytes: 2 });
|
||||
expect(preview.data.toString()).toBe("# ");
|
||||
expect(preview.truncated).toBe(true);
|
||||
await expect(svc.read(ws, "sub")).rejects.toMatchObject({ status: 400 });
|
||||
await expect(svc.read(ws, "nope.txt")).rejects.toMatchObject({ status: 404 });
|
||||
});
|
||||
@@ -86,6 +89,8 @@ describe("workspace-files-service", () => {
|
||||
|
||||
it("symlink escape: reads and writes are both rejected when the link points outside the Workspace", async () => {
|
||||
await fs.symlink(outside, path.join(ws, "link-out"));
|
||||
const root = await svc.list(ws, "");
|
||||
expect(root.entries.map((entry) => entry.name)).not.toContain("link-out");
|
||||
await expect(svc.list(ws, "link-out")).rejects.toMatchObject({ status: 400 });
|
||||
await expect(svc.read(ws, "link-out/secret.txt")).rejects.toMatchObject({ status: 400 });
|
||||
// Writing outside via a directory symlink: caught by the parent-directory realpath check.
|
||||
|
||||
Reference in New Issue
Block a user