feat(web,server,core): attach files to a message from the composer (#121)

Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Yaowei Zheng
2026-07-29 23:44:10 +08:00
committed by GitHub
parent 37fc715ce9
commit 1d23a7acaf
25 changed files with 1692 additions and 131 deletions
+12 -2
View File
@@ -604,11 +604,21 @@ export interface MessagesResponse {
// ---------------------------------------------------------------------------
/**
* A single Prompt's input parts: text or image (data: / http(s) URL).
* A single Prompt's input parts: text, image (data: / http(s) URL), or an uploaded file.
* Docs: /docs/server-api § "Session-Level Endpoints".
*/
export type TaskInputPart =
{ type: "text"; text: string } | { type: "image_url"; imageUrl: string };
| { type: "text"; text: string }
| { type: "image_url"; imageUrl: string }
/**
* File attachment (the composer's "+" menu): `dataUrl` is a base64 `data:` URL of the
* file's bytes, capped at 10MB each (413 `file_too_large` beyond that; the request as a
* whole still has to fit the global 20MB body limit). The server writes it into the
* Session scratchpad under a sanitized name and appends an `[attached file: <path>]` line
* to the message text — the bytes never enter the conversation, the model opens the file
* by path. `fileName` is the original name (no path separators, no `..`).
*/
| { type: "file"; fileName: string; dataUrl: string };
export interface TaskCreateRequest {
input: TaskInputPart[];
+18 -7
View File
@@ -11,6 +11,7 @@ import fsp from "node:fs/promises";
import path from "node:path";
import { Hono } from "hono";
import type { Context } from "hono";
import { bodyLimit } from "hono/body-limit";
import type { DatabaseSync } from "node:sqlite";
import type { ServerConfig } from "./config.js";
import { openDatabase } from "./db/database.js";
@@ -332,13 +333,23 @@ export function createApp(deps: AppDeps): Hono<AppEnv> {
}
// API common defenses: request body size cap (20MB) and write-request Content-Type (one of the CSRF MVP defenses).
app.use("/api/*", async (c, next) => {
const contentLength = Number(c.req.header("content-length") ?? 0);
if (contentLength > MAX_BODY_BYTES) {
throw new HttpError(413, "payload_too_large", "Request body exceeds the 20MB limit.");
}
await next();
});
//
// The cap has to be measured, not read: a chunked request carries no `content-length` at all,
// so a header check alone passes a body of any size — the sinks behind it (task input images,
// file attachments, Trace import) then decode whatever arrives. hono's bodyLimit keeps the
// header fast path when the length is declared and otherwise counts bytes off the stream,
// aborting the moment the total crosses the cap.
app.use(
"/api/*",
bodyLimit({
maxSize: MAX_BODY_BYTES,
// Its default is a bare text/plain 413; throw the App's own error instead so the response
// stays the documented `payload_too_large` body that every client already handles.
onError: () => {
throw new HttpError(413, "payload_too_large", "Request body exceeds the 20MB limit.");
},
}),
);
app.use("/api/*", jsonOnlyWrites);
// Public routes (no login required).
+134 -32
View File
@@ -46,6 +46,13 @@ import {
} from "../validate.js";
import type { AppDeps } from "../../app.js";
import { MAX_UPLOAD_BYTES } from "../../services/workspace-files-service.js";
import {
assertAttachmentBudget,
attachFilesToInput,
parseAttachmentPart,
removeAttachments,
} from "../../services/task-attachments.js";
import type { TaskAttachment } from "../../services/task-attachments.js";
/** Max title length for manual renames: looser than the auto-generated 30-char limit, to accommodate users' own organizing conventions. */
const SESSION_TITLE_MAX = 120;
@@ -72,13 +79,48 @@ const SESSION_CATEGORIES: readonly SessionCategory[] = [
"archived",
];
/** Validate Prompt input parts: text or image (data: / http(s) URL). */
function parseTaskInput(body: Record<string, unknown>): OmniMessage[] {
/**
* Resolve a scratchpad file name to an absolute path inside `dir`, or null when it could point
* anywhere else (the caller turns that into the same 404 a missing file gets, so a probe learns
* nothing either way).
*
* A character whitelist is deliberately NOT the guard: an attachment keeps the name the user
* gave it, `报告.pdf` included, so the check is structural instead — no separators, no control
* characters, not a relative marker — and then *confirmed* by resolving the path and requiring
* its parent to be this session's directory exactly. That last step is what actually contains
* the read: it also rejects the shapes a character class misses, such as a Windows
* drive-relative `C:evil.png`.
*/
function resolveScratchpadFile(dir: string, fileName: string): string | null {
if (!fileName || fileName === "." || fileName === "..") return null;
for (const ch of fileName) {
const code = ch.codePointAt(0)!;
if (code < 0x20 || code === 0x7f || ch === "/" || ch === "\\") return null;
}
const resolved = path.resolve(dir, fileName);
return path.dirname(resolved) === path.resolve(dir) ? resolved : null;
}
/**
* A validated Prompt: the message parts that go straight into the run, plus the file
* attachments, which still have to be written to disk (see attachFilesToInput). Kept apart
* because validation stays synchronous and side-effect free — nothing touches the filesystem
* until the request is known to be good, and goal mode can reject files before any bytes land.
*/
interface ParsedTaskInput {
messages: OmniMessage[];
attachments: TaskAttachment[];
}
/** Validate Prompt input parts: text, image (data: / http(s) URL), or an uploaded file. */
function parseTaskInput(body: Record<string, unknown>): ParsedTaskInput {
const input = body.input;
if (!Array.isArray(input) || input.length === 0) {
throw badRequest("input must be an array with at least one item.");
}
return input.map((item, i) => {
const messages: OmniMessage[] = [];
const attachments: TaskAttachment[] = [];
input.forEach((item, i) => {
if (item === null || typeof item !== "object" || Array.isArray(item)) {
throw badRequest(`input[${i}] must be an object.`);
}
@@ -87,7 +129,8 @@ function parseTaskInput(body: Record<string, unknown>): OmniMessage[] {
if (typeof part.text !== "string" || part.text.length === 0) {
throw badRequest(`input[${i}].text must be a non-empty string.`);
}
return userText(part.text);
messages.push(userText(part.text));
return;
}
if (part.type === "image_url") {
const url = part.imageUrl;
@@ -97,10 +140,21 @@ function parseTaskInput(body: Record<string, unknown>): OmniMessage[] {
) {
throw badRequest(`input[${i}].imageUrl only supports data: or http(s) URLs.`);
}
return imageUrlMessage(url);
messages.push(imageUrlMessage(url));
return;
}
throw badRequest(`input[${i}].type must be one of text / image_url.`);
if (part.type === "file") {
// Not an OmniMessage of its own: the file becomes an `[attached file: …]` line on the
// text message once written to the scratchpad, so it carries no payload into the run.
attachments.push(parseAttachmentPart(part, i));
// Per-request count / total-bytes caps, re-checked on every part so a hostile `input`
// is cut off at the item that crosses the line (see assertAttachmentBudget).
assertAttachmentBudget(attachments);
return;
}
throw badRequest(`input[${i}].type must be one of text / image_url / file.`);
});
return { messages, attachments };
}
/**
@@ -323,29 +377,33 @@ export function sessionsRoutes(deps: AppDeps): Hono<AppEnv> {
return c.body(null, 204);
});
// Session scratchpad files (e.g. input images saved to disk for image-unsupported
// models): read by filename, so the conversation UI can render a message's
// "[attached image: <path>]" attachment line back into an image. Restricted to this
// session's own scratchpad directory (the filename must not contain a path
// separator, blocking traversal); filenames include a timestamp and are globally
// unique, so the response is marked immutable and long-cacheable.
// Session scratchpad files (input images saved to disk for image-unsupported models, the
// composer's file attachments, model-generated temp files): read by filename, so the
// conversation UI can render a message's "[attached image: <path>]" attachment line back
// into an image. Restricted to this session's own scratchpad directory (see
// resolveScratchpadFile); a name is never reused for different bytes — uploads take a random
// suffix on collision — so the response is marked immutable and long-cacheable.
app.get("/:sessionId/scratchpad/:fileName", async (c) => {
const row = resolveSession(c);
const fileName = c.req.param("fileName") ?? "";
if (!/^[A-Za-z0-9._-]+$/.test(fileName) || fileName.includes("..")) {
throw new HttpError(404, "file_not_found", "File does not exist.");
}
const filePath = path.join(
scratchpadDir(deps.config.root, row.projectId, row.agentId),
row.sessionId,
const filePath = resolveScratchpadFile(
path.join(scratchpadDir(deps.config.root, row.projectId, row.agentId), row.sessionId),
fileName,
);
if (!filePath) throw new HttpError(404, "file_not_found", "File does not exist.");
let bytes: Buffer;
try {
bytes = await fs.readFile(filePath);
} catch {
throw new HttpError(404, "file_not_found", "File does not exist.");
}
// SECURITY BOUNDARY — do not extend casually. This map is an allowlist of types that are
// safe to hand a browser inline from the App's own origin, and it is the only reason the
// bytes below (arbitrary user uploads and Agent-written temp files) cannot become stored
// XSS. Every image type here is inert when rendered. Adding `.svg`, `.html`, `.pdf` or
// anything else that a browser parses as a document would look like a one-line convenience
// and would immediately be same-origin script execution — such a type needs the treatment
// the Workspace read gives it (plain-text downgrade or a sandbox CSP), not a map entry.
const MIME_BY_EXT: Record<string, string> = {
".png": "image/png",
".jpg": "image/jpeg",
@@ -353,9 +411,23 @@ export function sessionsRoutes(deps: AppDeps): Hono<AppEnv> {
".gif": "image/gif",
".webp": "image/webp",
};
const mime = MIME_BY_EXT[path.extname(fileName).toLowerCase()] ?? "application/octet-stream";
const mime = MIME_BY_EXT[path.extname(fileName).toLowerCase()];
return c.body(new Uint8Array(bytes), 200, {
"content-type": mime,
"content-type": mime ?? "application/octet-stream",
// nosniff: the composer's file attachments land in this same directory, so the bytes
// here are arbitrary user content served from the App's own origin — without it a
// browser could sniff an `application/octet-stream` upload back into HTML and run it
// same-origin (the same defense workspace file reads apply).
"x-content-type-options": "nosniff",
// Second, independent layer for everything that fell off the allowlist: the only reason
// this endpoint is fetched inline is the conversation's <img> tags, so anything that is
// not one of those images is served as a download and never renders as a document —
// nosniff alone would be the whole defense otherwise.
...(mime === undefined
? {
"content-disposition": `attachment; filename*=UTF-8''${encodeURIComponent(fileName)}`,
}
: {}),
"cache-control": "private, max-age=31536000, immutable",
});
});
@@ -420,33 +492,63 @@ export function sessionsRoutes(deps: AppDeps): Hono<AppEnv> {
const thinkingLevel = optionalEnum(body, "thinkingLevel", THINKING_LEVELS);
if (goal) {
// Goal mode: the input must be plain non-empty text (its marker-stripped text becomes
// the objective, re-injected every round — images have no place in the protocol).
const input = parseTaskInput(body);
const text = input
// the objective, re-injected every round — images and file attachments have no place in
// the protocol; rejected before any upload is written to disk).
const { messages, attachments } = parseTaskInput(body);
const text = messages
.filter((m) => (m.payload as { type?: string }).type === "text")
.map((m) => (m.payload as { text: string }).text)
.join("\n")
.trim();
if (!text || input.some((m) => (m.payload as { type?: string }).type !== "text")) {
if (
!text ||
attachments.length > 0 ||
messages.some((m) => (m.payload as { type?: string }).type !== "text")
) {
throw badRequest("goal mode requires text-only input (the objective).");
}
const { sessionId } = await deps.manager.startGoal(row.sessionId, {
input,
input: messages,
budget: goal.budget,
...(thinkingLevel !== undefined ? { thinkingLevel } : {}),
});
return c.json({ sessionId } satisfies TaskCreateResponse, 202);
}
const input = parseTaskInput(body);
const parsed = parseTaskInput(body);
// Follow-up queue: with queueIfBusy, a busy session enqueues the input instead of 409
// (auto-starts as an ordinary next task once idle; the response says which happened).
const queueIfBusy = body.queueIfBusy === true;
// 202: the Task executes on the server, decoupled from the SSE connection; sessionId is the current actual id (the new id after self-heal).
const { sessionId, queued } = await deps.manager.startTask(row.sessionId, input, {
...(thinkingLevel !== undefined ? { thinkingLevel } : {}),
queueIfBusy,
});
return c.json({ sessionId, queued } satisfies TaskCreateResponse, 202);
// Advisory pre-check, so the overwhelmingly common rejection — sending while a Task is
// running, without queueIfBusy — never writes bytes it would then have to take back. The
// authoritative check still runs under the Session lock inside startTask; this one is
// lock-free and may pass on a race, which the cleanup below covers.
deps.manager.assertCanAcceptTask(row.sessionId, { queueIfBusy });
// File attachments land in this Session's scratchpad (deleted along with the Session) and
// are handed to the model as `[attached file: <path>]` lines on the message text. Written
// even when the task ends up queued as a follow-up: the queued input must be complete, and
// the queue is drained by this same Session. A Trace-less Session that self-heals into a
// new id below keeps its files under the id they were written with — the paths in the
// message stay valid; only the delete-with-the-Session cleanup misses them in that case.
const { input, written } = await attachFilesToInput(
parsed.messages,
parsed.attachments,
scratchpadDir(deps.config.root, row.projectId, row.agentId),
row.sessionId,
);
try {
// 202: the Task executes on the server, decoupled from the SSE connection; sessionId is the current actual id (the new id after self-heal).
const { sessionId, queued } = await deps.manager.startTask(row.sessionId, input, {
...(thinkingLevel !== undefined ? { thinkingLevel } : {}),
queueIfBusy,
});
return c.json({ sessionId, queued } satisfies TaskCreateResponse, 202);
} catch (err) {
// The Task never started, so nothing references these files and nothing will ever clean
// them up — and the Web keeps the chips on failure, so the user's retry would otherwise
// land a second copy of every one of them.
await removeAttachments(written);
throw err;
}
});
// Mid-run steering: queue a user message for the running Task; core delivers it between
@@ -442,6 +442,24 @@ export class SessionManager {
// —— Task / compaction drive ——
/**
* Cheap, lock-free rehearsal of the 409/503 conditions startTask checks, throwing exactly the
* same HttpErrors. **Advisory only**: it neither takes the Session lock nor loads an entry, so
* a session that isn't in the active table reads as acceptable and a status change racing this
* call is not caught — the authoritative check is still the one inside startTask.
*
* It exists so a caller that has irreversible work to do first (POST /tasks writes the
* message's file attachments to disk) can find out about the ordinary "a Task is already
* running" rejection before doing it, instead of undoing it afterwards.
*/
assertCanAcceptTask(sessionId: string, opts?: { queueIfBusy?: boolean }): void {
this.assertOpen();
this.assertAgentNotDeleting(sessionId);
this.assertSessionNotDeleting(sessionId);
const entry = this.entries.get(sessionId);
if (entry && !opts?.queueIfBusy) this.assertIdle(entry);
}
/**
* Start a Task: get-or-load → 409
* mutual-exclusion check → publish the input messages first → drive run in the
@@ -0,0 +1,297 @@
/**
* Composer file attachments — the `{type:"file"}` variant of TaskInputPart.
*
* The browser has no Session while a chat is still a draft, so there is no upload endpoint to
* call before sending: the file rides the task request itself as a base64 `data:` URL, exactly
* like a pasted image does. On the way in it is written to the **Session scratchpad**
* (`<agent>/scratchpad/<sessionId>/`, deleted with the Session, so cleanup is free) and the
* message text gains one `[attached file: <absolute path>]` line per file — the bytes never
* enter the conversation, the model opens the file by path with its ordinary file tools.
*
* The line format and its placement are not defined here: they are shared with core's
* `[attached image: …]` producer and the Web renderer that parses both
* (`@prismshadow/penguin-core/markers` → attachment-lines.ts, plus `appendAttachmentLines`),
* so the two conventions cannot drift apart.
*/
import fs from "node:fs/promises";
import path from "node:path";
import { randomBytes } from "node:crypto";
import { appendAttachmentLines, attachedFileLine } from "@prismshadow/penguin-core";
import type { OmniMessage } from "@prismshadow/penguin-core";
import { HttpError } from "../http/errors.js";
import { badRequest } from "../http/validate.js";
/**
* Per-file cap. Deliberately below the workspace upload's 14MB: several attachments can ride
* one task request, and the whole body still has to fit the global 20MB limit (base64 inflates
* by 4/3), so a smaller per-file ceiling keeps "one big file" working while leaving room for
* "a handful of ordinary ones". Oversize is a 413, matching the workspace upload route.
*/
export const MAX_ATTACHMENT_BYTES = 10 * 1024 * 1024;
/**
* Per-request caps, checked while the parts are validated — before a single byte reaches the
* disk. They are deliberately NOT delegated to the global body cap: this module has to hold on
* its own, so that a change to that middleware (or a caller that never goes through it) cannot
* silently unbound it. Without them a legal 20MB body fits ~350k minimal `file` parts, which is
* 350k sequential writes into one directory and 350k marker lines on one message.
*
* 20 files is far past any plausible composer use — the chip row stops being usable long before
* that — while still allowing "drop a folder of small files in". 12MB of decoded bytes keeps one
* full-size 10MB attachment usable next to a couple of ordinary ones; base64 inflates by 4/3, so
* 12MB decoded is ~16MB of body and this cap, not the 20MB body cap, is the one a caller
* actually reaches.
*/
export const MAX_ATTACHMENT_COUNT = 20;
export const MAX_TOTAL_ATTACHMENT_BYTES = 12 * 1024 * 1024;
/**
* Longest stem kept on disk, measured in **UTF-8 bytes**: filesystems cap a name near 255
* bytes, and a CJK character costs three of them — a character count would let a Chinese name
* blow the real limit while an English one stayed far under it.
*/
const MAX_STEM_BYTES = 80;
/** Windows reserves these device names with or without an extension (`con`, `con.txt`), case-insensitively. */
const WINDOWS_RESERVED = /^(con|prn|aux|nul|com[1-9]|lpt[1-9])$/i;
/** Format and control characters (Unicode category C) — invisible, and the vector behind right-to-left file-name spoofing. */
const INVISIBLE_CHAR = /\p{C}/u;
/**
* True when a character must not reach a file name. ASCII keeps the long-standing whitelist:
* a space or a shell metacharacter inside a path the model is about to paste into a command is
* a footgun, so anything outside `[A-Za-z0-9._-]` still becomes `-` down there. Above ASCII the
* rule inverts — the character is kept as typed, so `报告.pdf` reaches the model as `报告.pdf`
* instead of collapsing to an anonymous `file.pdf` (CJK, accents and emoji are all harmless to
* a shell). The exception is Unicode category C: invisible controls, and the bidi overrides
* that let a name render as something it is not.
*/
function unsafeNameChar(ch: string): boolean {
if (/[A-Za-z0-9._-]/.test(ch)) return false;
return ch.codePointAt(0)! < 0x80 || INVISIBLE_CHAR.test(ch);
}
/** Replace every unsafe character with `-`, iterating code points so a surrogate pair survives intact. */
function sanitizeSegment(value: string): string {
return Array.from(value, (ch) => (unsafeNameChar(ch) ? "-" : ch)).join("");
}
/** Truncate to a UTF-8 byte budget on character boundaries (iterating a string yields whole code points, so a surrogate pair is never split). */
function truncateBytes(value: string, maxBytes: number): string {
if (Buffer.byteLength(value) <= maxBytes) return value;
let out = "";
for (const ch of value) {
if (Buffer.byteLength(out) + Buffer.byteLength(ch) > maxBytes) break;
out += ch;
}
return out;
}
/** The 6-hex space makes a second collision negligible; the cap only guards against a filesystem stuck on EEXIST. */
const MAX_NAME_ATTEMPTS = 16;
/** One validated attachment, bytes already decoded (they are held in memory only until the write below). */
export interface TaskAttachment {
/** Original file name as submitted (validated: non-empty, no path separators, no `..`). */
fileName: string;
bytes: Buffer;
}
/**
* Validate one `{type:"file"}` input part. Shape problems are 400s in the same style as the
* neighbouring text/image checks; only the size cap answers 413 (`file_too_large`, the code
* the Web App already has copy for). `index` is the part's position in `input`, so the message
* points at the offending item like the other input errors do.
*/
export function parseAttachmentPart(part: Record<string, unknown>, index: number): TaskAttachment {
const fileName = part.fileName;
// Path separators and `..` are rejected rather than sanitized away: the name is the user's,
// and a name that looks like a path means the caller is confused about the contract (the
// write below composes the path itself, and sanitization happens there).
if (
typeof fileName !== "string" ||
fileName.length === 0 ||
fileName.includes("/") ||
fileName.includes("\\") ||
fileName.includes("..") ||
fileName.includes("\0")
) {
throw badRequest(
`input[${index}].fileName must be a non-empty file name without path separators or "..".`,
);
}
const dataUrl = part.dataUrl;
// `[^,]*` for the media type, not `[^;,]*`: a browser may hand out parameters
// (`data:text/plain;charset=utf-8;base64,…`), and only the `;base64,` marker separates the
// type from the payload. The payload's character class is the actual check that it IS
// base64 (whitespace tolerated — line-wrapped encoders decode fine).
const match =
typeof dataUrl === "string" ? /^data:[^,]*;base64,([A-Za-z0-9+/=\s]+)$/.exec(dataUrl) : null;
if (!match) {
throw badRequest(`input[${index}].dataUrl must be a base64 data: URL of the file's bytes.`);
}
const bytes = Buffer.from(match[1]!, "base64");
if (bytes.length === 0) {
throw badRequest(`input[${index}].dataUrl decodes to an empty file.`);
}
if (bytes.length > MAX_ATTACHMENT_BYTES) {
throw new HttpError(
413,
"file_too_large",
`Attached file exceeds the ${MAX_ATTACHMENT_BYTES / (1024 * 1024)}MB limit.`,
);
}
return { fileName, bytes };
}
/**
* Enforce the per-request caps against everything accepted so far. Called after **each** `file`
* part rather than once at the end, so an oversized `input` stops at the part that crosses the
* line instead of base64-decoding the whole array first. Both answer 413: the count reuses a
* dedicated `too_many_files` code, the aggregate the `payload_too_large` the body cap already
* uses — from the caller's side it is the same "this request is too big" outcome.
*/
export function assertAttachmentBudget(attachments: TaskAttachment[]): void {
if (attachments.length > MAX_ATTACHMENT_COUNT) {
throw new HttpError(
413,
"too_many_files",
`A message may carry at most ${MAX_ATTACHMENT_COUNT} attached files.`,
);
}
let total = 0;
for (const a of attachments) total += a.bytes.length;
if (total > MAX_TOTAL_ATTACHMENT_BYTES) {
throw new HttpError(
413,
"payload_too_large",
`Attached files exceed the ${MAX_TOTAL_ATTACHMENT_BYTES / (1024 * 1024)}MB total limit for one message.`,
);
}
}
/**
* Map a submitted name onto a name that is safe on disk **and** still recognizably the user's
* own: `报告 2026.pdf` becomes `报告-2026.pdf` (see unsafeNameChar — the words survive, only the
* shell-hostile ASCII is replaced), so the model reads a meaningful path and a person looking
* at the message recognizes what they attached.
*
* The rest is Windows-shaped hygiene: trailing dots and spaces are dropped (Windows silently
* strips them, so `a.` and `a` would be the same file), a reserved device name is prefixed
* (`con.txt` → `_con.txt`), the stem is capped by UTF-8 bytes, and a stem that sanitizes away
* entirely falls back to `file` rather than producing a bare extension.
*/
function scratchpadName(fileName: string): string {
const ext = sanitizeSegment(path.extname(fileName));
const rawStem = fileName.slice(0, fileName.length - path.extname(fileName).length);
// Trim after truncating: a cut can expose a trailing dot or space that was mid-name before.
const stem = truncateBytes(sanitizeSegment(rawStem), MAX_STEM_BYTES).replace(/[. ]+$/, "");
// Nothing but replacement dashes carries no more information than an empty stem did.
if (!stem || /^-+$/.test(stem)) return `file${ext}`;
return WINDOWS_RESERVED.test(stem) ? `_${stem}${ext}` : `${stem}${ext}`;
}
/**
* Write one attachment into `dir` and return its absolute path. The plain sanitized name is
* tried first (the model — and the user reading the message — sees `report.pdf`, not an opaque
* id); "wx" makes the create exclusive, so a second upload of the same name lands next to the
* first as `report-3f9a1c.pdf` instead of overwriting it (same convention as core's image
* uploads).
*
* "wx" is O_CREAT|O_EXCL, which also refuses to follow a symlink at the final component: a link
* planted at `report.pdf` fails with EEXIST and the retry allocates a suffixed name instead of
* writing through it. The containment check for the *directory* is separate — see openScratchpadDir.
*/
async function writeAttachment(dir: string, attachment: TaskAttachment): Promise<string> {
const base = scratchpadName(attachment.fileName);
const ext = path.extname(base);
const stem = base.slice(0, base.length - ext.length);
for (let attempt = 0; attempt < MAX_NAME_ATTEMPTS; attempt++) {
const name = attempt === 0 ? base : `${stem}-${randomBytes(3).toString("hex")}${ext}`;
const file = path.join(dir, name);
try {
await fs.writeFile(file, attachment.bytes, { flag: "wx" });
return file;
} catch (err) {
if ((err as NodeJS.ErrnoException).code !== "EEXIST") throw err;
}
}
throw new Error(
`failed to allocate a unique attachment file name under ${dir} after ${MAX_NAME_ATTEMPTS} attempts`,
);
}
/**
* Create (or reuse) this Session's scratchpad directory and hand back the path to write into.
*
* `fs.mkdir(…, {recursive:true})` succeeds silently when the directory is already a **symlink**
* to somewhere else, and nothing downstream would notice — so the result is realpath'd and
* required to still sit inside the Agent's scratchpad root, the same containment rule the
* Workspace upload path applies (workspace-files-service.resolveWriteParent). Only the Agent
* process can plant such a link and it runs as this server's uid, so this is consistency rather
* than a privilege boundary; it costs one resolution per message that carries attachments.
*
* The check is on the canonical path but the write stays on the logical one: the path travels
* into the message text, and the read/delete endpoints address a Session by its logical
* directory, so canonicalizing here would only make those disagree on hosts where the data root
* itself sits behind a link (macOS `/var`, a Windows 8.3 temp path).
*/
async function openScratchpadDir(root: string, sessionId: string): Promise<string> {
const dir = path.join(root, sessionId);
await fs.mkdir(dir, { recursive: true });
const canonicalRoot = await fs.realpath(root);
const rel = path.relative(canonicalRoot, await fs.realpath(dir));
if (rel !== sessionId) {
throw new Error(
`session scratchpad ${dir} resolves outside the agent scratchpad root; refusing to write attachments`,
);
}
return dir;
}
/** Best-effort undo of a batch of writes; errors are swallowed because every caller is already on an error path (a failed cleanup must not replace the original failure). */
export async function removeAttachments(files: string[]): Promise<void> {
await Promise.all(files.map((f) => fs.rm(f, { force: true }).catch(() => {})));
}
/** Result of a write batch: the Prompt to run, plus the paths written so the caller can undo them if the Task never starts. */
export interface AttachedFiles {
input: OmniMessage[];
written: string[];
}
/**
* Land every attachment in the Session scratchpad under `root` and return the Prompt with one
* `[attached file: <path>]` line appended per file. Placement follows core's shared rule
* (after the last user text message; attachments-only input becomes a line-only text
* message), so a Prompt carrying both images and files still ends in a single trailing block.
* Returns `messages` untouched when there is nothing to attach — no directory is created.
*
* All-or-nothing: a failure part-way through the batch removes what it already wrote, so a 500
* never leaves files on disk that no message refers to. The caller owns the other half of that
* guarantee — if starting the Task fails afterwards it must call removeAttachments(written),
* otherwise the user's retry would land a second copy of every file.
*/
export async function attachFilesToInput(
messages: OmniMessage[],
attachments: TaskAttachment[],
root: string,
sessionId: string,
): Promise<AttachedFiles> {
if (attachments.length === 0) return { input: messages, written: [] };
const dir = await openScratchpadDir(root, sessionId);
const written: string[] = [];
try {
// Sequential on purpose: the exclusive-create retry above resolves collisions against files
// that already exist, and writing the batch one at a time keeps two same-named uploads in
// the same message from racing each other for the plain name.
for (const attachment of attachments) {
written.push(await writeAttachment(dir, attachment));
}
} catch (err) {
await removeAttachments(written);
throw err;
}
return { input: appendAttachmentLines(messages, written.map(attachedFileLine)), written };
}