feat(web,server,core): attach files to a message from the composer (#121)
Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -604,11 +604,21 @@ export interface MessagesResponse {
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* A single Prompt's input parts: text or image (data: / http(s) URL).
|
||||
* A single Prompt's input parts: text, image (data: / http(s) URL), or an uploaded file.
|
||||
* Docs: /docs/server-api § "Session-Level Endpoints".
|
||||
*/
|
||||
export type TaskInputPart =
|
||||
{ type: "text"; text: string } | { type: "image_url"; imageUrl: string };
|
||||
| { type: "text"; text: string }
|
||||
| { type: "image_url"; imageUrl: string }
|
||||
/**
|
||||
* File attachment (the composer's "+" menu): `dataUrl` is a base64 `data:` URL of the
|
||||
* file's bytes, capped at 10MB each (413 `file_too_large` beyond that; the request as a
|
||||
* whole still has to fit the global 20MB body limit). The server writes it into the
|
||||
* Session scratchpad under a sanitized name and appends an `[attached file: <path>]` line
|
||||
* to the message text — the bytes never enter the conversation, the model opens the file
|
||||
* by path. `fileName` is the original name (no path separators, no `..`).
|
||||
*/
|
||||
| { type: "file"; fileName: string; dataUrl: string };
|
||||
|
||||
export interface TaskCreateRequest {
|
||||
input: TaskInputPart[];
|
||||
|
||||
@@ -11,6 +11,7 @@ import fsp from "node:fs/promises";
|
||||
import path from "node:path";
|
||||
import { Hono } from "hono";
|
||||
import type { Context } from "hono";
|
||||
import { bodyLimit } from "hono/body-limit";
|
||||
import type { DatabaseSync } from "node:sqlite";
|
||||
import type { ServerConfig } from "./config.js";
|
||||
import { openDatabase } from "./db/database.js";
|
||||
@@ -332,13 +333,23 @@ export function createApp(deps: AppDeps): Hono<AppEnv> {
|
||||
}
|
||||
|
||||
// API common defenses: request body size cap (20MB) and write-request Content-Type (one of the CSRF MVP defenses).
|
||||
app.use("/api/*", async (c, next) => {
|
||||
const contentLength = Number(c.req.header("content-length") ?? 0);
|
||||
if (contentLength > MAX_BODY_BYTES) {
|
||||
throw new HttpError(413, "payload_too_large", "Request body exceeds the 20MB limit.");
|
||||
}
|
||||
await next();
|
||||
});
|
||||
//
|
||||
// The cap has to be measured, not read: a chunked request carries no `content-length` at all,
|
||||
// so a header check alone passes a body of any size — the sinks behind it (task input images,
|
||||
// file attachments, Trace import) then decode whatever arrives. hono's bodyLimit keeps the
|
||||
// header fast path when the length is declared and otherwise counts bytes off the stream,
|
||||
// aborting the moment the total crosses the cap.
|
||||
app.use(
|
||||
"/api/*",
|
||||
bodyLimit({
|
||||
maxSize: MAX_BODY_BYTES,
|
||||
// Its default is a bare text/plain 413; throw the App's own error instead so the response
|
||||
// stays the documented `payload_too_large` body that every client already handles.
|
||||
onError: () => {
|
||||
throw new HttpError(413, "payload_too_large", "Request body exceeds the 20MB limit.");
|
||||
},
|
||||
}),
|
||||
);
|
||||
app.use("/api/*", jsonOnlyWrites);
|
||||
|
||||
// Public routes (no login required).
|
||||
|
||||
@@ -46,6 +46,13 @@ import {
|
||||
} from "../validate.js";
|
||||
import type { AppDeps } from "../../app.js";
|
||||
import { MAX_UPLOAD_BYTES } from "../../services/workspace-files-service.js";
|
||||
import {
|
||||
assertAttachmentBudget,
|
||||
attachFilesToInput,
|
||||
parseAttachmentPart,
|
||||
removeAttachments,
|
||||
} from "../../services/task-attachments.js";
|
||||
import type { TaskAttachment } from "../../services/task-attachments.js";
|
||||
|
||||
/** Max title length for manual renames: looser than the auto-generated 30-char limit, to accommodate users' own organizing conventions. */
|
||||
const SESSION_TITLE_MAX = 120;
|
||||
@@ -72,13 +79,48 @@ const SESSION_CATEGORIES: readonly SessionCategory[] = [
|
||||
"archived",
|
||||
];
|
||||
|
||||
/** Validate Prompt input parts: text or image (data: / http(s) URL). */
|
||||
function parseTaskInput(body: Record<string, unknown>): OmniMessage[] {
|
||||
/**
|
||||
* Resolve a scratchpad file name to an absolute path inside `dir`, or null when it could point
|
||||
* anywhere else (the caller turns that into the same 404 a missing file gets, so a probe learns
|
||||
* nothing either way).
|
||||
*
|
||||
* A character whitelist is deliberately NOT the guard: an attachment keeps the name the user
|
||||
* gave it, `报告.pdf` included, so the check is structural instead — no separators, no control
|
||||
* characters, not a relative marker — and then *confirmed* by resolving the path and requiring
|
||||
* its parent to be this session's directory exactly. That last step is what actually contains
|
||||
* the read: it also rejects the shapes a character class misses, such as a Windows
|
||||
* drive-relative `C:evil.png`.
|
||||
*/
|
||||
function resolveScratchpadFile(dir: string, fileName: string): string | null {
|
||||
if (!fileName || fileName === "." || fileName === "..") return null;
|
||||
for (const ch of fileName) {
|
||||
const code = ch.codePointAt(0)!;
|
||||
if (code < 0x20 || code === 0x7f || ch === "/" || ch === "\\") return null;
|
||||
}
|
||||
const resolved = path.resolve(dir, fileName);
|
||||
return path.dirname(resolved) === path.resolve(dir) ? resolved : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* A validated Prompt: the message parts that go straight into the run, plus the file
|
||||
* attachments, which still have to be written to disk (see attachFilesToInput). Kept apart
|
||||
* because validation stays synchronous and side-effect free — nothing touches the filesystem
|
||||
* until the request is known to be good, and goal mode can reject files before any bytes land.
|
||||
*/
|
||||
interface ParsedTaskInput {
|
||||
messages: OmniMessage[];
|
||||
attachments: TaskAttachment[];
|
||||
}
|
||||
|
||||
/** Validate Prompt input parts: text, image (data: / http(s) URL), or an uploaded file. */
|
||||
function parseTaskInput(body: Record<string, unknown>): ParsedTaskInput {
|
||||
const input = body.input;
|
||||
if (!Array.isArray(input) || input.length === 0) {
|
||||
throw badRequest("input must be an array with at least one item.");
|
||||
}
|
||||
return input.map((item, i) => {
|
||||
const messages: OmniMessage[] = [];
|
||||
const attachments: TaskAttachment[] = [];
|
||||
input.forEach((item, i) => {
|
||||
if (item === null || typeof item !== "object" || Array.isArray(item)) {
|
||||
throw badRequest(`input[${i}] must be an object.`);
|
||||
}
|
||||
@@ -87,7 +129,8 @@ function parseTaskInput(body: Record<string, unknown>): OmniMessage[] {
|
||||
if (typeof part.text !== "string" || part.text.length === 0) {
|
||||
throw badRequest(`input[${i}].text must be a non-empty string.`);
|
||||
}
|
||||
return userText(part.text);
|
||||
messages.push(userText(part.text));
|
||||
return;
|
||||
}
|
||||
if (part.type === "image_url") {
|
||||
const url = part.imageUrl;
|
||||
@@ -97,10 +140,21 @@ function parseTaskInput(body: Record<string, unknown>): OmniMessage[] {
|
||||
) {
|
||||
throw badRequest(`input[${i}].imageUrl only supports data: or http(s) URLs.`);
|
||||
}
|
||||
return imageUrlMessage(url);
|
||||
messages.push(imageUrlMessage(url));
|
||||
return;
|
||||
}
|
||||
throw badRequest(`input[${i}].type must be one of text / image_url.`);
|
||||
if (part.type === "file") {
|
||||
// Not an OmniMessage of its own: the file becomes an `[attached file: …]` line on the
|
||||
// text message once written to the scratchpad, so it carries no payload into the run.
|
||||
attachments.push(parseAttachmentPart(part, i));
|
||||
// Per-request count / total-bytes caps, re-checked on every part so a hostile `input`
|
||||
// is cut off at the item that crosses the line (see assertAttachmentBudget).
|
||||
assertAttachmentBudget(attachments);
|
||||
return;
|
||||
}
|
||||
throw badRequest(`input[${i}].type must be one of text / image_url / file.`);
|
||||
});
|
||||
return { messages, attachments };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -323,29 +377,33 @@ export function sessionsRoutes(deps: AppDeps): Hono<AppEnv> {
|
||||
return c.body(null, 204);
|
||||
});
|
||||
|
||||
// Session scratchpad files (e.g. input images saved to disk for image-unsupported
|
||||
// models): read by filename, so the conversation UI can render a message's
|
||||
// "[attached image: <path>]" attachment line back into an image. Restricted to this
|
||||
// session's own scratchpad directory (the filename must not contain a path
|
||||
// separator, blocking traversal); filenames include a timestamp and are globally
|
||||
// unique, so the response is marked immutable and long-cacheable.
|
||||
// Session scratchpad files (input images saved to disk for image-unsupported models, the
|
||||
// composer's file attachments, model-generated temp files): read by filename, so the
|
||||
// conversation UI can render a message's "[attached image: <path>]" attachment line back
|
||||
// into an image. Restricted to this session's own scratchpad directory (see
|
||||
// resolveScratchpadFile); a name is never reused for different bytes — uploads take a random
|
||||
// suffix on collision — so the response is marked immutable and long-cacheable.
|
||||
app.get("/:sessionId/scratchpad/:fileName", async (c) => {
|
||||
const row = resolveSession(c);
|
||||
const fileName = c.req.param("fileName") ?? "";
|
||||
if (!/^[A-Za-z0-9._-]+$/.test(fileName) || fileName.includes("..")) {
|
||||
throw new HttpError(404, "file_not_found", "File does not exist.");
|
||||
}
|
||||
const filePath = path.join(
|
||||
scratchpadDir(deps.config.root, row.projectId, row.agentId),
|
||||
row.sessionId,
|
||||
const filePath = resolveScratchpadFile(
|
||||
path.join(scratchpadDir(deps.config.root, row.projectId, row.agentId), row.sessionId),
|
||||
fileName,
|
||||
);
|
||||
if (!filePath) throw new HttpError(404, "file_not_found", "File does not exist.");
|
||||
let bytes: Buffer;
|
||||
try {
|
||||
bytes = await fs.readFile(filePath);
|
||||
} catch {
|
||||
throw new HttpError(404, "file_not_found", "File does not exist.");
|
||||
}
|
||||
// SECURITY BOUNDARY — do not extend casually. This map is an allowlist of types that are
|
||||
// safe to hand a browser inline from the App's own origin, and it is the only reason the
|
||||
// bytes below (arbitrary user uploads and Agent-written temp files) cannot become stored
|
||||
// XSS. Every image type here is inert when rendered. Adding `.svg`, `.html`, `.pdf` or
|
||||
// anything else that a browser parses as a document would look like a one-line convenience
|
||||
// and would immediately be same-origin script execution — such a type needs the treatment
|
||||
// the Workspace read gives it (plain-text downgrade or a sandbox CSP), not a map entry.
|
||||
const MIME_BY_EXT: Record<string, string> = {
|
||||
".png": "image/png",
|
||||
".jpg": "image/jpeg",
|
||||
@@ -353,9 +411,23 @@ export function sessionsRoutes(deps: AppDeps): Hono<AppEnv> {
|
||||
".gif": "image/gif",
|
||||
".webp": "image/webp",
|
||||
};
|
||||
const mime = MIME_BY_EXT[path.extname(fileName).toLowerCase()] ?? "application/octet-stream";
|
||||
const mime = MIME_BY_EXT[path.extname(fileName).toLowerCase()];
|
||||
return c.body(new Uint8Array(bytes), 200, {
|
||||
"content-type": mime,
|
||||
"content-type": mime ?? "application/octet-stream",
|
||||
// nosniff: the composer's file attachments land in this same directory, so the bytes
|
||||
// here are arbitrary user content served from the App's own origin — without it a
|
||||
// browser could sniff an `application/octet-stream` upload back into HTML and run it
|
||||
// same-origin (the same defense workspace file reads apply).
|
||||
"x-content-type-options": "nosniff",
|
||||
// Second, independent layer for everything that fell off the allowlist: the only reason
|
||||
// this endpoint is fetched inline is the conversation's <img> tags, so anything that is
|
||||
// not one of those images is served as a download and never renders as a document —
|
||||
// nosniff alone would be the whole defense otherwise.
|
||||
...(mime === undefined
|
||||
? {
|
||||
"content-disposition": `attachment; filename*=UTF-8''${encodeURIComponent(fileName)}`,
|
||||
}
|
||||
: {}),
|
||||
"cache-control": "private, max-age=31536000, immutable",
|
||||
});
|
||||
});
|
||||
@@ -420,33 +492,63 @@ export function sessionsRoutes(deps: AppDeps): Hono<AppEnv> {
|
||||
const thinkingLevel = optionalEnum(body, "thinkingLevel", THINKING_LEVELS);
|
||||
if (goal) {
|
||||
// Goal mode: the input must be plain non-empty text (its marker-stripped text becomes
|
||||
// the objective, re-injected every round — images have no place in the protocol).
|
||||
const input = parseTaskInput(body);
|
||||
const text = input
|
||||
// the objective, re-injected every round — images and file attachments have no place in
|
||||
// the protocol; rejected before any upload is written to disk).
|
||||
const { messages, attachments } = parseTaskInput(body);
|
||||
const text = messages
|
||||
.filter((m) => (m.payload as { type?: string }).type === "text")
|
||||
.map((m) => (m.payload as { text: string }).text)
|
||||
.join("\n")
|
||||
.trim();
|
||||
if (!text || input.some((m) => (m.payload as { type?: string }).type !== "text")) {
|
||||
if (
|
||||
!text ||
|
||||
attachments.length > 0 ||
|
||||
messages.some((m) => (m.payload as { type?: string }).type !== "text")
|
||||
) {
|
||||
throw badRequest("goal mode requires text-only input (the objective).");
|
||||
}
|
||||
const { sessionId } = await deps.manager.startGoal(row.sessionId, {
|
||||
input,
|
||||
input: messages,
|
||||
budget: goal.budget,
|
||||
...(thinkingLevel !== undefined ? { thinkingLevel } : {}),
|
||||
});
|
||||
return c.json({ sessionId } satisfies TaskCreateResponse, 202);
|
||||
}
|
||||
const input = parseTaskInput(body);
|
||||
const parsed = parseTaskInput(body);
|
||||
// Follow-up queue: with queueIfBusy, a busy session enqueues the input instead of 409
|
||||
// (auto-starts as an ordinary next task once idle; the response says which happened).
|
||||
const queueIfBusy = body.queueIfBusy === true;
|
||||
// 202: the Task executes on the server, decoupled from the SSE connection; sessionId is the current actual id (the new id after self-heal).
|
||||
const { sessionId, queued } = await deps.manager.startTask(row.sessionId, input, {
|
||||
...(thinkingLevel !== undefined ? { thinkingLevel } : {}),
|
||||
queueIfBusy,
|
||||
});
|
||||
return c.json({ sessionId, queued } satisfies TaskCreateResponse, 202);
|
||||
// Advisory pre-check, so the overwhelmingly common rejection — sending while a Task is
|
||||
// running, without queueIfBusy — never writes bytes it would then have to take back. The
|
||||
// authoritative check still runs under the Session lock inside startTask; this one is
|
||||
// lock-free and may pass on a race, which the cleanup below covers.
|
||||
deps.manager.assertCanAcceptTask(row.sessionId, { queueIfBusy });
|
||||
// File attachments land in this Session's scratchpad (deleted along with the Session) and
|
||||
// are handed to the model as `[attached file: <path>]` lines on the message text. Written
|
||||
// even when the task ends up queued as a follow-up: the queued input must be complete, and
|
||||
// the queue is drained by this same Session. A Trace-less Session that self-heals into a
|
||||
// new id below keeps its files under the id they were written with — the paths in the
|
||||
// message stay valid; only the delete-with-the-Session cleanup misses them in that case.
|
||||
const { input, written } = await attachFilesToInput(
|
||||
parsed.messages,
|
||||
parsed.attachments,
|
||||
scratchpadDir(deps.config.root, row.projectId, row.agentId),
|
||||
row.sessionId,
|
||||
);
|
||||
try {
|
||||
// 202: the Task executes on the server, decoupled from the SSE connection; sessionId is the current actual id (the new id after self-heal).
|
||||
const { sessionId, queued } = await deps.manager.startTask(row.sessionId, input, {
|
||||
...(thinkingLevel !== undefined ? { thinkingLevel } : {}),
|
||||
queueIfBusy,
|
||||
});
|
||||
return c.json({ sessionId, queued } satisfies TaskCreateResponse, 202);
|
||||
} catch (err) {
|
||||
// The Task never started, so nothing references these files and nothing will ever clean
|
||||
// them up — and the Web keeps the chips on failure, so the user's retry would otherwise
|
||||
// land a second copy of every one of them.
|
||||
await removeAttachments(written);
|
||||
throw err;
|
||||
}
|
||||
});
|
||||
|
||||
// Mid-run steering: queue a user message for the running Task; core delivers it between
|
||||
|
||||
@@ -442,6 +442,24 @@ export class SessionManager {
|
||||
|
||||
// —— Task / compaction drive ——
|
||||
|
||||
/**
|
||||
* Cheap, lock-free rehearsal of the 409/503 conditions startTask checks, throwing exactly the
|
||||
* same HttpErrors. **Advisory only**: it neither takes the Session lock nor loads an entry, so
|
||||
* a session that isn't in the active table reads as acceptable and a status change racing this
|
||||
* call is not caught — the authoritative check is still the one inside startTask.
|
||||
*
|
||||
* It exists so a caller that has irreversible work to do first (POST /tasks writes the
|
||||
* message's file attachments to disk) can find out about the ordinary "a Task is already
|
||||
* running" rejection before doing it, instead of undoing it afterwards.
|
||||
*/
|
||||
assertCanAcceptTask(sessionId: string, opts?: { queueIfBusy?: boolean }): void {
|
||||
this.assertOpen();
|
||||
this.assertAgentNotDeleting(sessionId);
|
||||
this.assertSessionNotDeleting(sessionId);
|
||||
const entry = this.entries.get(sessionId);
|
||||
if (entry && !opts?.queueIfBusy) this.assertIdle(entry);
|
||||
}
|
||||
|
||||
/**
|
||||
* Start a Task: get-or-load → 409
|
||||
* mutual-exclusion check → publish the input messages first → drive run in the
|
||||
|
||||
@@ -0,0 +1,297 @@
|
||||
/**
|
||||
* Composer file attachments — the `{type:"file"}` variant of TaskInputPart.
|
||||
*
|
||||
* The browser has no Session while a chat is still a draft, so there is no upload endpoint to
|
||||
* call before sending: the file rides the task request itself as a base64 `data:` URL, exactly
|
||||
* like a pasted image does. On the way in it is written to the **Session scratchpad**
|
||||
* (`<agent>/scratchpad/<sessionId>/`, deleted with the Session, so cleanup is free) and the
|
||||
* message text gains one `[attached file: <absolute path>]` line per file — the bytes never
|
||||
* enter the conversation, the model opens the file by path with its ordinary file tools.
|
||||
*
|
||||
* The line format and its placement are not defined here: they are shared with core's
|
||||
* `[attached image: …]` producer and the Web renderer that parses both
|
||||
* (`@prismshadow/penguin-core/markers` → attachment-lines.ts, plus `appendAttachmentLines`),
|
||||
* so the two conventions cannot drift apart.
|
||||
*/
|
||||
import fs from "node:fs/promises";
|
||||
import path from "node:path";
|
||||
import { randomBytes } from "node:crypto";
|
||||
import { appendAttachmentLines, attachedFileLine } from "@prismshadow/penguin-core";
|
||||
import type { OmniMessage } from "@prismshadow/penguin-core";
|
||||
import { HttpError } from "../http/errors.js";
|
||||
import { badRequest } from "../http/validate.js";
|
||||
|
||||
/**
|
||||
* Per-file cap. Deliberately below the workspace upload's 14MB: several attachments can ride
|
||||
* one task request, and the whole body still has to fit the global 20MB limit (base64 inflates
|
||||
* by 4/3), so a smaller per-file ceiling keeps "one big file" working while leaving room for
|
||||
* "a handful of ordinary ones". Oversize is a 413, matching the workspace upload route.
|
||||
*/
|
||||
export const MAX_ATTACHMENT_BYTES = 10 * 1024 * 1024;
|
||||
|
||||
/**
|
||||
* Per-request caps, checked while the parts are validated — before a single byte reaches the
|
||||
* disk. They are deliberately NOT delegated to the global body cap: this module has to hold on
|
||||
* its own, so that a change to that middleware (or a caller that never goes through it) cannot
|
||||
* silently unbound it. Without them a legal 20MB body fits ~350k minimal `file` parts, which is
|
||||
* 350k sequential writes into one directory and 350k marker lines on one message.
|
||||
*
|
||||
* 20 files is far past any plausible composer use — the chip row stops being usable long before
|
||||
* that — while still allowing "drop a folder of small files in". 12MB of decoded bytes keeps one
|
||||
* full-size 10MB attachment usable next to a couple of ordinary ones; base64 inflates by 4/3, so
|
||||
* 12MB decoded is ~16MB of body and this cap, not the 20MB body cap, is the one a caller
|
||||
* actually reaches.
|
||||
*/
|
||||
export const MAX_ATTACHMENT_COUNT = 20;
|
||||
export const MAX_TOTAL_ATTACHMENT_BYTES = 12 * 1024 * 1024;
|
||||
|
||||
/**
|
||||
* Longest stem kept on disk, measured in **UTF-8 bytes**: filesystems cap a name near 255
|
||||
* bytes, and a CJK character costs three of them — a character count would let a Chinese name
|
||||
* blow the real limit while an English one stayed far under it.
|
||||
*/
|
||||
const MAX_STEM_BYTES = 80;
|
||||
|
||||
/** Windows reserves these device names with or without an extension (`con`, `con.txt`), case-insensitively. */
|
||||
const WINDOWS_RESERVED = /^(con|prn|aux|nul|com[1-9]|lpt[1-9])$/i;
|
||||
|
||||
/** Format and control characters (Unicode category C) — invisible, and the vector behind right-to-left file-name spoofing. */
|
||||
const INVISIBLE_CHAR = /\p{C}/u;
|
||||
|
||||
/**
|
||||
* True when a character must not reach a file name. ASCII keeps the long-standing whitelist:
|
||||
* a space or a shell metacharacter inside a path the model is about to paste into a command is
|
||||
* a footgun, so anything outside `[A-Za-z0-9._-]` still becomes `-` down there. Above ASCII the
|
||||
* rule inverts — the character is kept as typed, so `报告.pdf` reaches the model as `报告.pdf`
|
||||
* instead of collapsing to an anonymous `file.pdf` (CJK, accents and emoji are all harmless to
|
||||
* a shell). The exception is Unicode category C: invisible controls, and the bidi overrides
|
||||
* that let a name render as something it is not.
|
||||
*/
|
||||
function unsafeNameChar(ch: string): boolean {
|
||||
if (/[A-Za-z0-9._-]/.test(ch)) return false;
|
||||
return ch.codePointAt(0)! < 0x80 || INVISIBLE_CHAR.test(ch);
|
||||
}
|
||||
|
||||
/** Replace every unsafe character with `-`, iterating code points so a surrogate pair survives intact. */
|
||||
function sanitizeSegment(value: string): string {
|
||||
return Array.from(value, (ch) => (unsafeNameChar(ch) ? "-" : ch)).join("");
|
||||
}
|
||||
|
||||
/** Truncate to a UTF-8 byte budget on character boundaries (iterating a string yields whole code points, so a surrogate pair is never split). */
|
||||
function truncateBytes(value: string, maxBytes: number): string {
|
||||
if (Buffer.byteLength(value) <= maxBytes) return value;
|
||||
let out = "";
|
||||
for (const ch of value) {
|
||||
if (Buffer.byteLength(out) + Buffer.byteLength(ch) > maxBytes) break;
|
||||
out += ch;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** The 6-hex space makes a second collision negligible; the cap only guards against a filesystem stuck on EEXIST. */
|
||||
const MAX_NAME_ATTEMPTS = 16;
|
||||
|
||||
/** One validated attachment, bytes already decoded (they are held in memory only until the write below). */
|
||||
export interface TaskAttachment {
|
||||
/** Original file name as submitted (validated: non-empty, no path separators, no `..`). */
|
||||
fileName: string;
|
||||
bytes: Buffer;
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate one `{type:"file"}` input part. Shape problems are 400s in the same style as the
|
||||
* neighbouring text/image checks; only the size cap answers 413 (`file_too_large`, the code
|
||||
* the Web App already has copy for). `index` is the part's position in `input`, so the message
|
||||
* points at the offending item like the other input errors do.
|
||||
*/
|
||||
export function parseAttachmentPart(part: Record<string, unknown>, index: number): TaskAttachment {
|
||||
const fileName = part.fileName;
|
||||
// Path separators and `..` are rejected rather than sanitized away: the name is the user's,
|
||||
// and a name that looks like a path means the caller is confused about the contract (the
|
||||
// write below composes the path itself, and sanitization happens there).
|
||||
if (
|
||||
typeof fileName !== "string" ||
|
||||
fileName.length === 0 ||
|
||||
fileName.includes("/") ||
|
||||
fileName.includes("\\") ||
|
||||
fileName.includes("..") ||
|
||||
fileName.includes("\0")
|
||||
) {
|
||||
throw badRequest(
|
||||
`input[${index}].fileName must be a non-empty file name without path separators or "..".`,
|
||||
);
|
||||
}
|
||||
const dataUrl = part.dataUrl;
|
||||
// `[^,]*` for the media type, not `[^;,]*`: a browser may hand out parameters
|
||||
// (`data:text/plain;charset=utf-8;base64,…`), and only the `;base64,` marker separates the
|
||||
// type from the payload. The payload's character class is the actual check that it IS
|
||||
// base64 (whitespace tolerated — line-wrapped encoders decode fine).
|
||||
const match =
|
||||
typeof dataUrl === "string" ? /^data:[^,]*;base64,([A-Za-z0-9+/=\s]+)$/.exec(dataUrl) : null;
|
||||
if (!match) {
|
||||
throw badRequest(`input[${index}].dataUrl must be a base64 data: URL of the file's bytes.`);
|
||||
}
|
||||
const bytes = Buffer.from(match[1]!, "base64");
|
||||
if (bytes.length === 0) {
|
||||
throw badRequest(`input[${index}].dataUrl decodes to an empty file.`);
|
||||
}
|
||||
if (bytes.length > MAX_ATTACHMENT_BYTES) {
|
||||
throw new HttpError(
|
||||
413,
|
||||
"file_too_large",
|
||||
`Attached file exceeds the ${MAX_ATTACHMENT_BYTES / (1024 * 1024)}MB limit.`,
|
||||
);
|
||||
}
|
||||
return { fileName, bytes };
|
||||
}
|
||||
|
||||
/**
|
||||
* Enforce the per-request caps against everything accepted so far. Called after **each** `file`
|
||||
* part rather than once at the end, so an oversized `input` stops at the part that crosses the
|
||||
* line instead of base64-decoding the whole array first. Both answer 413: the count reuses a
|
||||
* dedicated `too_many_files` code, the aggregate the `payload_too_large` the body cap already
|
||||
* uses — from the caller's side it is the same "this request is too big" outcome.
|
||||
*/
|
||||
export function assertAttachmentBudget(attachments: TaskAttachment[]): void {
|
||||
if (attachments.length > MAX_ATTACHMENT_COUNT) {
|
||||
throw new HttpError(
|
||||
413,
|
||||
"too_many_files",
|
||||
`A message may carry at most ${MAX_ATTACHMENT_COUNT} attached files.`,
|
||||
);
|
||||
}
|
||||
let total = 0;
|
||||
for (const a of attachments) total += a.bytes.length;
|
||||
if (total > MAX_TOTAL_ATTACHMENT_BYTES) {
|
||||
throw new HttpError(
|
||||
413,
|
||||
"payload_too_large",
|
||||
`Attached files exceed the ${MAX_TOTAL_ATTACHMENT_BYTES / (1024 * 1024)}MB total limit for one message.`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Map a submitted name onto a name that is safe on disk **and** still recognizably the user's
|
||||
* own: `报告 2026.pdf` becomes `报告-2026.pdf` (see unsafeNameChar — the words survive, only the
|
||||
* shell-hostile ASCII is replaced), so the model reads a meaningful path and a person looking
|
||||
* at the message recognizes what they attached.
|
||||
*
|
||||
* The rest is Windows-shaped hygiene: trailing dots and spaces are dropped (Windows silently
|
||||
* strips them, so `a.` and `a` would be the same file), a reserved device name is prefixed
|
||||
* (`con.txt` → `_con.txt`), the stem is capped by UTF-8 bytes, and a stem that sanitizes away
|
||||
* entirely falls back to `file` rather than producing a bare extension.
|
||||
*/
|
||||
function scratchpadName(fileName: string): string {
|
||||
const ext = sanitizeSegment(path.extname(fileName));
|
||||
const rawStem = fileName.slice(0, fileName.length - path.extname(fileName).length);
|
||||
// Trim after truncating: a cut can expose a trailing dot or space that was mid-name before.
|
||||
const stem = truncateBytes(sanitizeSegment(rawStem), MAX_STEM_BYTES).replace(/[. ]+$/, "");
|
||||
// Nothing but replacement dashes carries no more information than an empty stem did.
|
||||
if (!stem || /^-+$/.test(stem)) return `file${ext}`;
|
||||
return WINDOWS_RESERVED.test(stem) ? `_${stem}${ext}` : `${stem}${ext}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write one attachment into `dir` and return its absolute path. The plain sanitized name is
|
||||
* tried first (the model — and the user reading the message — sees `report.pdf`, not an opaque
|
||||
* id); "wx" makes the create exclusive, so a second upload of the same name lands next to the
|
||||
* first as `report-3f9a1c.pdf` instead of overwriting it (same convention as core's image
|
||||
* uploads).
|
||||
*
|
||||
* "wx" is O_CREAT|O_EXCL, which also refuses to follow a symlink at the final component: a link
|
||||
* planted at `report.pdf` fails with EEXIST and the retry allocates a suffixed name instead of
|
||||
* writing through it. The containment check for the *directory* is separate — see openScratchpadDir.
|
||||
*/
|
||||
async function writeAttachment(dir: string, attachment: TaskAttachment): Promise<string> {
|
||||
const base = scratchpadName(attachment.fileName);
|
||||
const ext = path.extname(base);
|
||||
const stem = base.slice(0, base.length - ext.length);
|
||||
for (let attempt = 0; attempt < MAX_NAME_ATTEMPTS; attempt++) {
|
||||
const name = attempt === 0 ? base : `${stem}-${randomBytes(3).toString("hex")}${ext}`;
|
||||
const file = path.join(dir, name);
|
||||
try {
|
||||
await fs.writeFile(file, attachment.bytes, { flag: "wx" });
|
||||
return file;
|
||||
} catch (err) {
|
||||
if ((err as NodeJS.ErrnoException).code !== "EEXIST") throw err;
|
||||
}
|
||||
}
|
||||
throw new Error(
|
||||
`failed to allocate a unique attachment file name under ${dir} after ${MAX_NAME_ATTEMPTS} attempts`,
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create (or reuse) this Session's scratchpad directory and hand back the path to write into.
|
||||
*
|
||||
* `fs.mkdir(…, {recursive:true})` succeeds silently when the directory is already a **symlink**
|
||||
* to somewhere else, and nothing downstream would notice — so the result is realpath'd and
|
||||
* required to still sit inside the Agent's scratchpad root, the same containment rule the
|
||||
* Workspace upload path applies (workspace-files-service.resolveWriteParent). Only the Agent
|
||||
* process can plant such a link and it runs as this server's uid, so this is consistency rather
|
||||
* than a privilege boundary; it costs one resolution per message that carries attachments.
|
||||
*
|
||||
* The check is on the canonical path but the write stays on the logical one: the path travels
|
||||
* into the message text, and the read/delete endpoints address a Session by its logical
|
||||
* directory, so canonicalizing here would only make those disagree on hosts where the data root
|
||||
* itself sits behind a link (macOS `/var`, a Windows 8.3 temp path).
|
||||
*/
|
||||
async function openScratchpadDir(root: string, sessionId: string): Promise<string> {
|
||||
const dir = path.join(root, sessionId);
|
||||
await fs.mkdir(dir, { recursive: true });
|
||||
const canonicalRoot = await fs.realpath(root);
|
||||
const rel = path.relative(canonicalRoot, await fs.realpath(dir));
|
||||
if (rel !== sessionId) {
|
||||
throw new Error(
|
||||
`session scratchpad ${dir} resolves outside the agent scratchpad root; refusing to write attachments`,
|
||||
);
|
||||
}
|
||||
return dir;
|
||||
}
|
||||
|
||||
/** Best-effort undo of a batch of writes; errors are swallowed because every caller is already on an error path (a failed cleanup must not replace the original failure). */
|
||||
export async function removeAttachments(files: string[]): Promise<void> {
|
||||
await Promise.all(files.map((f) => fs.rm(f, { force: true }).catch(() => {})));
|
||||
}
|
||||
|
||||
/** Result of a write batch: the Prompt to run, plus the paths written so the caller can undo them if the Task never starts. */
|
||||
export interface AttachedFiles {
|
||||
input: OmniMessage[];
|
||||
written: string[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Land every attachment in the Session scratchpad under `root` and return the Prompt with one
|
||||
* `[attached file: <path>]` line appended per file. Placement follows core's shared rule
|
||||
* (after the last user text message; attachments-only input becomes a line-only text
|
||||
* message), so a Prompt carrying both images and files still ends in a single trailing block.
|
||||
* Returns `messages` untouched when there is nothing to attach — no directory is created.
|
||||
*
|
||||
* All-or-nothing: a failure part-way through the batch removes what it already wrote, so a 500
|
||||
* never leaves files on disk that no message refers to. The caller owns the other half of that
|
||||
* guarantee — if starting the Task fails afterwards it must call removeAttachments(written),
|
||||
* otherwise the user's retry would land a second copy of every file.
|
||||
*/
|
||||
export async function attachFilesToInput(
|
||||
messages: OmniMessage[],
|
||||
attachments: TaskAttachment[],
|
||||
root: string,
|
||||
sessionId: string,
|
||||
): Promise<AttachedFiles> {
|
||||
if (attachments.length === 0) return { input: messages, written: [] };
|
||||
const dir = await openScratchpadDir(root, sessionId);
|
||||
const written: string[] = [];
|
||||
try {
|
||||
// Sequential on purpose: the exclusive-create retry above resolves collisions against files
|
||||
// that already exist, and writing the batch one at a time keeps two same-named uploads in
|
||||
// the same message from racing each other for the plain name.
|
||||
for (const attachment of attachments) {
|
||||
written.push(await writeAttachment(dir, attachment));
|
||||
}
|
||||
} catch (err) {
|
||||
await removeAttachments(written);
|
||||
throw err;
|
||||
}
|
||||
return { input: appendAttachmentLines(messages, written.map(attachedFileLine)), written };
|
||||
}
|
||||
Reference in New Issue
Block a user