feat(core,tooling): Windows support — shell selection, install.ps1, win-x64 release package, Windows CI (#79)
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -126,6 +126,7 @@ describe("App Data Dir / Agent ID placeholders", () => {
|
||||
modelId: "deepseek-v4-pro",
|
||||
platform: "linux",
|
||||
osVersion: "test",
|
||||
shell: "bash",
|
||||
date: "2026-07-08",
|
||||
});
|
||||
expect(prompt).toContain("Agent ID: env_agent");
|
||||
|
||||
@@ -132,8 +132,10 @@ describe("ContextEngine ReAct loop (mock LLM, approve callback)", () => {
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
await rm(workspace, { recursive: true, force: true });
|
||||
await rm(traces, { recursive: true, force: true });
|
||||
// Retries (here and in the other cleanups below): on Windows a just-killed process tree
|
||||
// releases its cwd locks asynchronously, so an immediate rm can hit EBUSY.
|
||||
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
await rm(traces, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
});
|
||||
|
||||
it("approves a tool call, writes the file, returns the final answer, traces it", async () => {
|
||||
@@ -733,7 +735,7 @@ describe("ContextEngine async/incremental tool calls (overlapping execution)", (
|
||||
workspace = await mkdtemp(join(tmpdir(), "penguin-ws2-"));
|
||||
});
|
||||
afterEach(async () => {
|
||||
await rm(workspace, { recursive: true, force: true });
|
||||
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
});
|
||||
|
||||
it("emits both tool calls in one round; second is approved while the first executes; outputs come back in completion order", async () => {
|
||||
@@ -809,7 +811,13 @@ describe("ContextEngine async/incremental tool calls (overlapping execution)", (
|
||||
// command has not finished yet) -- i.e., execution does not block the next approval.
|
||||
expect(approvedAt["t2"]!).toBeLessThan(firstCompleteAt["t1"] ?? Infinity);
|
||||
// The fast b.txt finishes first, the slow a.txt finishes later (outputs in completion order).
|
||||
expect(firstCompleteAt["t2"]!).toBeLessThan(firstCompleteAt["t1"]!);
|
||||
// POSIX only: on Windows a cold Git-Bash spawn costs 1-2s, which can swamp the 400ms sleep
|
||||
// delta that makes t1 "the slow one" — CI has seen the two complete within 5ms — so the
|
||||
// relative completion order is not controllable there. The overlap assertion above and the
|
||||
// file contents below still run on Windows.
|
||||
if (process.platform !== "win32") {
|
||||
expect(firstCompleteAt["t2"]!).toBeLessThan(firstCompleteAt["t1"]!);
|
||||
}
|
||||
|
||||
expect(await readFile(join(workspace, "a.txt"), "utf8")).toBe("one");
|
||||
expect(await readFile(join(workspace, "b.txt"), "utf8")).toBe("two");
|
||||
@@ -895,7 +903,7 @@ describe("ContextEngine tool execution resilience", () => {
|
||||
workspace = await mkdtemp(join(tmpdir(), "penguin-ws4-"));
|
||||
});
|
||||
afterEach(async () => {
|
||||
await rm(workspace, { recursive: true, force: true });
|
||||
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
});
|
||||
|
||||
it("feeds a failed tool output back and keeps tool_use/result paired (Environment converges errors, never throws)", async () => {
|
||||
@@ -1105,7 +1113,7 @@ describe("ContextEngine abort during execution", () => {
|
||||
workspace = await mkdtemp(join(tmpdir(), "penguin-ws3-"));
|
||||
});
|
||||
afterEach(async () => {
|
||||
await rm(workspace, { recursive: true, force: true });
|
||||
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
});
|
||||
|
||||
it("aborting a long-running tool ends the turn, emits abort, and carries tool results over (model output completed)", async () => {
|
||||
@@ -1255,7 +1263,7 @@ describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => {
|
||||
workspace = await mkdtemp(join(tmpdir(), "penguin-ws4-"));
|
||||
});
|
||||
afterEach(async () => {
|
||||
await rm(workspace, { recursive: true, force: true });
|
||||
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
});
|
||||
|
||||
it("auto-retries on LLM timeout: original input + [turn_retried] carrying partial products", async () => {
|
||||
@@ -1792,8 +1800,8 @@ describe("ContextEngine mid-run steering ([user_steering])", () => {
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
await rm(workspace, { recursive: true, force: true });
|
||||
await rm(traces, { recursive: true, force: true });
|
||||
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
await rm(traces, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
});
|
||||
|
||||
/** Fake environment: streams a delta then closes with a fixed complete output (no real shell). */
|
||||
|
||||
@@ -67,7 +67,9 @@ beforeEach(async () => {
|
||||
afterEach(async () => {
|
||||
if (originalHome === undefined) delete process.env.HOME;
|
||||
else process.env.HOME = originalHome;
|
||||
await rm(tmp, { recursive: true, force: true });
|
||||
// Retries: on Windows a just-killed process tree releases its cwd/file locks asynchronously,
|
||||
// so an immediate recursive rm can hit EBUSY; fs.rm retries those with a linear backoff.
|
||||
await rm(tmp, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
});
|
||||
|
||||
describe("Environment.listTools", () => {
|
||||
@@ -485,9 +487,12 @@ describe("Environment.executeTool — relaxed tool contract", () => {
|
||||
|
||||
describe("Environment.executeTool — timeoutMs (PRN-013)", () => {
|
||||
it("fails a tool exceeding timeoutMs, keeps prior output, and streams the timeout reason", async () => {
|
||||
// The timeout must stay below MIN_YIELD_MS (250): for larger values exec_command yields to
|
||||
// background (with a process_id) before the Environment timeout can ever fire.
|
||||
const timeoutMs = 200;
|
||||
const env = new Environment({
|
||||
workspaceDir: tmp,
|
||||
toolConfig: makeToolConfig(execTool({ timeoutMs: 200 })),
|
||||
toolConfig: makeToolConfig(execTool({ timeoutMs })),
|
||||
});
|
||||
const startedAt = Date.now();
|
||||
|
||||
@@ -513,8 +518,13 @@ describe("Environment.executeTool — timeoutMs (PRN-013)", () => {
|
||||
};
|
||||
expect(last.type).toBe("tool_call_output");
|
||||
expect(last.stop_reason).toBe("failed");
|
||||
expect(last.output).toContain("begin");
|
||||
expect(last.output).toContain("[tool timeout: exceeded 200ms]");
|
||||
// Kept-prior-output is asserted only where the shell can win the race: a Git-Bash login
|
||||
// shell on Windows needs several hundred ms to start, so nothing is printed before a
|
||||
// sub-250ms timeout there — the timeout mechanics above are still fully exercised.
|
||||
if (process.platform !== "win32") {
|
||||
expect(last.output).toContain("begin");
|
||||
}
|
||||
expect(last.output).toContain(`[tool timeout: exceeded ${timeoutMs}ms]`);
|
||||
// The timeout marker is also produced via streaming: concatenating the streamed deltas ==
|
||||
// the complete content.
|
||||
const streamed = messages
|
||||
@@ -735,14 +745,14 @@ describe("Environment.executeTool — robustness", () => {
|
||||
describe("Environment.toolPermission", () => {
|
||||
it("returns the configured permission for a known tool", () => {
|
||||
const env = new Environment({
|
||||
workspaceDir: "/tmp",
|
||||
workspaceDir: tmpdir(),
|
||||
toolConfig: makeToolConfig(execTool({ permission: "rw" })),
|
||||
});
|
||||
expect(env.toolPermission("exec_command")).toBe("rw");
|
||||
});
|
||||
|
||||
it("returns undefined for an unknown tool", () => {
|
||||
const env = new Environment({ workspaceDir: "/tmp", toolConfig: makeToolConfig() });
|
||||
const env = new Environment({ workspaceDir: tmpdir(), toolConfig: makeToolConfig() });
|
||||
expect(env.toolPermission("nope")).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -91,7 +91,9 @@ beforeEach(async () => {
|
||||
|
||||
afterEach(async () => {
|
||||
env.dispose();
|
||||
await rm(tmp, { recursive: true, force: true });
|
||||
// Retries: on Windows a just-killed process tree releases its cwd/file locks asynchronously,
|
||||
// so an immediate recursive rm can hit EBUSY; fs.rm retries those with a linear backoff.
|
||||
await rm(tmp, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
});
|
||||
|
||||
describe("exec_command — long-running command sessions", () => {
|
||||
|
||||
@@ -514,15 +514,19 @@ describe("edit_file — review follow-ups", () => {
|
||||
describe("write_file — review follow-ups", () => {
|
||||
const tool = () => createWriteFileTool(def(WRITE_FILE_NAME, "rw"));
|
||||
|
||||
it("preserves the permission bits of an overwritten file (atomic rename)", async () => {
|
||||
const file = path.join(tmp, "mode.txt");
|
||||
await writeFile(file, "old");
|
||||
await chmod(file, 0o600);
|
||||
const { result } = await run(tool(), { file_path: "mode.txt", content: "new" }, tmp);
|
||||
expect(result?.stopReason).toBeUndefined();
|
||||
expect((await stat(file)).mode & 0o777).toBe(0o600);
|
||||
expect(await readFile(file, "utf8")).toBe("new");
|
||||
});
|
||||
// POSIX-only: Windows has no owner-only mode bits to preserve (chmod maps to the read-only attribute).
|
||||
it.skipIf(process.platform === "win32")(
|
||||
"preserves the permission bits of an overwritten file (atomic rename)",
|
||||
async () => {
|
||||
const file = path.join(tmp, "mode.txt");
|
||||
await writeFile(file, "old");
|
||||
await chmod(file, 0o600);
|
||||
const { result } = await run(tool(), { file_path: "mode.txt", content: "new" }, tmp);
|
||||
expect(result?.stopReason).toBeUndefined();
|
||||
expect((await stat(file)).mode & 0o777).toBe(0o600);
|
||||
expect(await readFile(file, "utf8")).toBe("new");
|
||||
},
|
||||
);
|
||||
|
||||
it("writes atomically: no temp files are left behind", async () => {
|
||||
await run(tool(), { file_path: "fresh.txt", content: "x" }, tmp);
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
/**
|
||||
* Unit tests for the command-session shell resolver (pure function; platform, env and the
|
||||
* PATH probe are injected — no real shells are spawned here).
|
||||
*/
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { resolveShell } from "../src/environment/tools/command/shell.js";
|
||||
|
||||
const POWERSHELL_ARGS = ["-NoLogo", "-NoProfile", "-Command"];
|
||||
|
||||
/** A whichAll stub resolving only the given names (value = returned PATH matches). */
|
||||
function which(table: Record<string, string[]>): (cmd: string) => string[] {
|
||||
return (cmd) => table[cmd] ?? [];
|
||||
}
|
||||
|
||||
describe("resolveShell — POSIX", () => {
|
||||
it("uses bash -lc on linux without probing (today's behavior, unchanged)", () => {
|
||||
let probed = false;
|
||||
const shell = resolveShell({
|
||||
platform: "linux",
|
||||
env: {},
|
||||
whichAll: () => {
|
||||
probed = true;
|
||||
return [];
|
||||
},
|
||||
});
|
||||
expect(shell).toEqual({ command: "bash", args: ["-lc"], name: "bash" });
|
||||
expect(probed).toBe(false);
|
||||
});
|
||||
|
||||
it("uses bash -lc on darwin", () => {
|
||||
const shell = resolveShell({ platform: "darwin", env: {} });
|
||||
expect(shell).toEqual({ command: "bash", args: ["-lc"], name: "bash" });
|
||||
});
|
||||
});
|
||||
|
||||
describe("resolveShell — win32 probing", () => {
|
||||
it("prefers bash on PATH (Git for Windows)", () => {
|
||||
const shell = resolveShell({
|
||||
platform: "win32",
|
||||
env: {},
|
||||
whichAll: which({
|
||||
bash: ["C:\\Program Files\\Git\\bin\\bash.exe"],
|
||||
pwsh: ["C:\\Program Files\\PowerShell\\7\\pwsh.exe"],
|
||||
}),
|
||||
});
|
||||
expect(shell).toEqual({ command: "bash", args: ["-lc"], name: "bash" });
|
||||
});
|
||||
|
||||
it("skips the WSL launcher bash under the system root and falls through to pwsh", () => {
|
||||
const shell = resolveShell({
|
||||
platform: "win32",
|
||||
env: { SystemRoot: "C:\\WINDOWS" },
|
||||
whichAll: which({
|
||||
bash: ["C:\\Windows\\System32\\bash.exe"],
|
||||
pwsh: ["C:\\Program Files\\PowerShell\\7\\pwsh.exe"],
|
||||
}),
|
||||
});
|
||||
expect(shell).toEqual({ command: "pwsh", args: POWERSHELL_ARGS, name: "pwsh" });
|
||||
});
|
||||
|
||||
it("falls back to pwsh when bash is absent", () => {
|
||||
const shell = resolveShell({
|
||||
platform: "win32",
|
||||
env: {},
|
||||
whichAll: which({ pwsh: ["C:\\Program Files\\PowerShell\\7\\pwsh.exe"] }),
|
||||
});
|
||||
expect(shell).toEqual({ command: "pwsh", args: POWERSHELL_ARGS, name: "pwsh" });
|
||||
});
|
||||
|
||||
it("falls back to powershell when neither bash nor pwsh resolve", () => {
|
||||
const shell = resolveShell({ platform: "win32", env: {}, whichAll: which({}) });
|
||||
expect(shell).toEqual({ command: "powershell", args: POWERSHELL_ARGS, name: "powershell" });
|
||||
});
|
||||
});
|
||||
|
||||
describe("resolveShell — PENGUIN_SHELL override", () => {
|
||||
it("wins on every platform and keeps POSIX-style args for a POSIX shell path", () => {
|
||||
const shell = resolveShell({
|
||||
platform: "linux",
|
||||
env: { PENGUIN_SHELL: "/usr/bin/zsh" },
|
||||
});
|
||||
expect(shell).toEqual({ command: "/usr/bin/zsh", args: ["-lc"], name: "zsh" });
|
||||
});
|
||||
|
||||
it("uses PowerShell-style args when the basename is pwsh (case/extension-insensitive)", () => {
|
||||
const shell = resolveShell({
|
||||
platform: "win32",
|
||||
env: { PENGUIN_SHELL: "C:\\Program Files\\PowerShell\\7\\pwsh.EXE" },
|
||||
whichAll: which({ bash: ["C:\\Program Files\\Git\\bin\\bash.exe"] }),
|
||||
});
|
||||
expect(shell).toEqual({
|
||||
command: "C:\\Program Files\\PowerShell\\7\\pwsh.EXE",
|
||||
args: POWERSHELL_ARGS,
|
||||
name: "pwsh",
|
||||
});
|
||||
});
|
||||
|
||||
it("uses PowerShell-style args for a bare powershell name", () => {
|
||||
const shell = resolveShell({ platform: "win32", env: { PENGUIN_SHELL: "powershell" } });
|
||||
expect(shell).toEqual({ command: "powershell", args: POWERSHELL_ARGS, name: "powershell" });
|
||||
});
|
||||
|
||||
it("uses cmd-style args when the basename is cmd", () => {
|
||||
const shell = resolveShell({ platform: "win32", env: { PENGUIN_SHELL: "cmd" } });
|
||||
expect(shell).toEqual({ command: "cmd", args: ["/d", "/s", "/c"], name: "cmd" });
|
||||
});
|
||||
|
||||
it("ignores a blank PENGUIN_SHELL", () => {
|
||||
const shell = resolveShell({ platform: "linux", env: { PENGUIN_SHELL: " " } });
|
||||
expect(shell).toEqual({ command: "bash", args: ["-lc"], name: "bash" });
|
||||
});
|
||||
});
|
||||
@@ -16,6 +16,7 @@ import {
|
||||
PLATFORM_PLACEHOLDER,
|
||||
PROJECT_DIR_PLACEHOLDER,
|
||||
SESSION_ID_PLACEHOLDER,
|
||||
SHELL_PLACEHOLDER,
|
||||
addModel,
|
||||
setVisionModel,
|
||||
agentsMdPath,
|
||||
@@ -64,7 +65,10 @@ afterEach(async () => {
|
||||
} else {
|
||||
process.env.PENGUIN_HOME = prevHome;
|
||||
}
|
||||
await fs.rm(tmpRoot, { recursive: true, force: true });
|
||||
// Retries: when a test times out, vitest runs this cleanup while the test's un-cancelled
|
||||
// init may still be writing files, so an immediate recursive rm can hit ENOTEMPTY on
|
||||
// Windows (fs.rm retries ENOTEMPTY/EBUSY/EPERM); a no-op when removal succeeds first try.
|
||||
await fs.rm(tmpRoot, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
|
||||
});
|
||||
|
||||
async function exists(p: string): Promise<boolean> {
|
||||
@@ -83,87 +87,96 @@ describe("paths / resolveRoot", () => {
|
||||
});
|
||||
|
||||
describe("loadOrInitAgentState", () => {
|
||||
it("initializes an empty agent directory with the full state layout", async () => {
|
||||
const state = await loadOrInitAgentState();
|
||||
expect(state.root).toBe(tmpRoot);
|
||||
expect(state.projectId).toBe(DEFAULT_PROJECT_ID);
|
||||
expect(state.agentId).toBe(DEFAULT_AGENT_ID);
|
||||
// Timeout: initialization writes the full layout — 15 library skills plus the example
|
||||
// benchmark, dozens of small files — and this first init test also pays the cold-I/O cost
|
||||
// (first-touch reads of the skills package, Defender scans) on Windows runners, where a
|
||||
// slow-disk moment has pushed it past the 5s default. Purely a failure deadline: passing
|
||||
// runs stay as fast as before on every platform.
|
||||
it(
|
||||
"initializes an empty agent directory with the full state layout",
|
||||
{ timeout: 20_000 },
|
||||
async () => {
|
||||
const state = await loadOrInitAgentState();
|
||||
expect(state.root).toBe(tmpRoot);
|
||||
expect(state.projectId).toBe(DEFAULT_PROJECT_ID);
|
||||
expect(state.agentId).toBe(DEFAULT_AGENT_ID);
|
||||
|
||||
const root = tmpRoot;
|
||||
expect(await exists(systemConfigPath(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
expect(await exists(agentsMdPath(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
expect(await exists(toolsDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
expect(await exists(memoryDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
expect(await exists(skillsDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
// The scratchpad/ directory alongside agent_state (model temp files get a subdirectory per Session id).
|
||||
expect(await exists(scratchpadDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
const root = tmpRoot;
|
||||
expect(await exists(systemConfigPath(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
expect(await exists(agentsMdPath(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
expect(await exists(toolsDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
expect(await exists(memoryDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
expect(await exists(skillsDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
// The scratchpad/ directory alongside agent_state (model temp files get a subdirectory per Session id).
|
||||
expect(await exists(scratchpadDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
|
||||
|
||||
expect(state.stateDir).toBe(agentStateDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID));
|
||||
expect(state.stateDir).toBe(agentStateDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID));
|
||||
|
||||
// The default system Prompt states the Agent's identity, without repeating tool details
|
||||
// already in the tool schema (Suggested workflows only points to the run_subagent
|
||||
// delegation entry point).
|
||||
expect(state.systemConfig.system_prompt).toContain("PenguinHarness");
|
||||
expect(state.systemConfig.system_prompt).not.toContain("exec_command");
|
||||
// Suggested workflows absorbs Subagent delegation and task conventions (self-reported
|
||||
// identity as a soft convention, parallelism, file exchange).
|
||||
expect(state.systemConfig.system_prompt).toContain("# Suggested workflows");
|
||||
expect(state.systemConfig.system_prompt).toContain("run_subagent");
|
||||
expect(state.systemConfig.system_prompt).toContain("Caller agent");
|
||||
// The default AGENTS.md is empty: it carries no preset guidance.
|
||||
expect(state.agentsMd).toBe("");
|
||||
expect(state.systemConfig.system_prompt).toContain(AGENTS_MD_PLACEHOLDER);
|
||||
expect(state.systemConfig.system_prompt).toContain(SESSION_ID_PLACEHOLDER);
|
||||
expect(state.systemConfig.system_prompt).toContain(CWD_PLACEHOLDER);
|
||||
expect(state.systemConfig.system_prompt).toContain(PLATFORM_PLACEHOLDER);
|
||||
expect(state.systemConfig.system_prompt).toContain(OS_VERSION_PLACEHOLDER);
|
||||
expect(state.systemConfig.system_prompt).toContain(DATE_PLACEHOLDER);
|
||||
// AGENTS.md and the Environment injection sit at the end of the template, with AGENTS.md
|
||||
// before Environment; the [developer_instructions] wrapper text is written directly into
|
||||
// the template (the Prompt is transparent about the config).
|
||||
expect(state.systemConfig.system_prompt).toContain("[developer_instructions]");
|
||||
expect(state.systemConfig.system_prompt).toContain("[/developer_instructions]");
|
||||
// The default template explains the semantics of system-synthesized markers to the model,
|
||||
// and recommends preferring tool use.
|
||||
expect(state.systemConfig.system_prompt).toContain("[turn_aborted]");
|
||||
expect(state.systemConfig.system_prompt).toContain("[turn_retried]");
|
||||
expect(state.systemConfig.system_prompt).toContain("[context_summary]");
|
||||
expect(state.systemConfig.system_prompt).toContain("[user_steering]");
|
||||
expect(state.systemConfig.system_prompt).toContain("# Tool use");
|
||||
// Privacy hardening: explicitly forbids reading .project_config.toml (the sole config file,
|
||||
// which holds API keys) and each Agent's .vault.toml, and states that config can only be
|
||||
// changed via the CLI (penguin config ...).
|
||||
expect(state.systemConfig.system_prompt).toContain("Never read");
|
||||
expect(state.systemConfig.system_prompt).toContain(".project_config.toml");
|
||||
expect(state.systemConfig.system_prompt).toContain("agent_state/.vault.toml");
|
||||
expect(state.systemConfig.system_prompt).toContain("CLI-only");
|
||||
expect(state.systemConfig.system_prompt).toContain("penguin config");
|
||||
expect(state.systemConfig.system_prompt).not.toContain(".credentials.toml");
|
||||
expect(state.systemConfig.system_prompt.indexOf(AGENTS_MD_PLACEHOLDER)).toBeLessThan(
|
||||
state.systemConfig.system_prompt.indexOf("# Environment"),
|
||||
);
|
||||
// The # Vault and # Skills body sections plus their placeholders: the default template
|
||||
// places them after [/developer_instructions] and before # Environment, in the order
|
||||
// Vault -> Skills (the statement text is part of the template body, kept even with no
|
||||
// keys/skills).
|
||||
const tpl = state.systemConfig.system_prompt;
|
||||
expect(tpl).toContain("# Vault");
|
||||
expect(tpl).toContain(VAULT_KEYS_PLACEHOLDER);
|
||||
expect(tpl).toContain("# Skills");
|
||||
expect(tpl).toContain(SKILL_METADATA_PLACEHOLDER);
|
||||
expect(tpl).toContain("[use_skills]");
|
||||
expect(tpl.indexOf("[/developer_instructions]")).toBeLessThan(tpl.indexOf("# Vault"));
|
||||
expect(tpl.indexOf("# Vault")).toBeLessThan(tpl.indexOf(VAULT_KEYS_PLACEHOLDER));
|
||||
expect(tpl.indexOf(VAULT_KEYS_PLACEHOLDER)).toBeLessThan(tpl.indexOf("# Skills"));
|
||||
expect(tpl.indexOf("# Skills")).toBeLessThan(tpl.indexOf(SKILL_METADATA_PLACEHOLDER));
|
||||
expect(tpl.indexOf(SKILL_METADATA_PLACEHOLDER)).toBeLessThan(tpl.indexOf("# Environment"));
|
||||
expect(state.systemConfig.model?.max_tokens).toBe(32000);
|
||||
expect(state.systemConfig.model?.thinking_level).toBe("medium");
|
||||
expect(state.systemConfig.model?.timeoutMs).toBe(120000);
|
||||
expect(state.systemConfig.tools?.mcpServers).toEqual([]);
|
||||
expect(Object.hasOwn(state.systemConfig, "description")).toBe(false);
|
||||
expect(Object.hasOwn(state.systemConfig, "subagents")).toBe(false);
|
||||
});
|
||||
// The default system Prompt states the Agent's identity, without repeating tool details
|
||||
// already in the tool schema (Suggested workflows only points to the run_subagent
|
||||
// delegation entry point).
|
||||
expect(state.systemConfig.system_prompt).toContain("PenguinHarness");
|
||||
expect(state.systemConfig.system_prompt).not.toContain("exec_command");
|
||||
// Suggested workflows absorbs Subagent delegation and task conventions (self-reported
|
||||
// identity as a soft convention, parallelism, file exchange).
|
||||
expect(state.systemConfig.system_prompt).toContain("# Suggested workflows");
|
||||
expect(state.systemConfig.system_prompt).toContain("run_subagent");
|
||||
expect(state.systemConfig.system_prompt).toContain("Caller agent");
|
||||
// The default AGENTS.md is empty: it carries no preset guidance.
|
||||
expect(state.agentsMd).toBe("");
|
||||
expect(state.systemConfig.system_prompt).toContain(AGENTS_MD_PLACEHOLDER);
|
||||
expect(state.systemConfig.system_prompt).toContain(SESSION_ID_PLACEHOLDER);
|
||||
expect(state.systemConfig.system_prompt).toContain(CWD_PLACEHOLDER);
|
||||
expect(state.systemConfig.system_prompt).toContain(PLATFORM_PLACEHOLDER);
|
||||
expect(state.systemConfig.system_prompt).toContain(OS_VERSION_PLACEHOLDER);
|
||||
expect(state.systemConfig.system_prompt).toContain(DATE_PLACEHOLDER);
|
||||
// AGENTS.md and the Environment injection sit at the end of the template, with AGENTS.md
|
||||
// before Environment; the [developer_instructions] wrapper text is written directly into
|
||||
// the template (the Prompt is transparent about the config).
|
||||
expect(state.systemConfig.system_prompt).toContain("[developer_instructions]");
|
||||
expect(state.systemConfig.system_prompt).toContain("[/developer_instructions]");
|
||||
// The default template explains the semantics of system-synthesized markers to the model,
|
||||
// and recommends preferring tool use.
|
||||
expect(state.systemConfig.system_prompt).toContain("[turn_aborted]");
|
||||
expect(state.systemConfig.system_prompt).toContain("[turn_retried]");
|
||||
expect(state.systemConfig.system_prompt).toContain("[context_summary]");
|
||||
expect(state.systemConfig.system_prompt).toContain("[user_steering]");
|
||||
expect(state.systemConfig.system_prompt).toContain("# Tool use");
|
||||
// Privacy hardening: explicitly forbids reading .project_config.toml (the sole config file,
|
||||
// which holds API keys) and each Agent's .vault.toml, and states that config can only be
|
||||
// changed via the CLI (penguin config ...).
|
||||
expect(state.systemConfig.system_prompt).toContain("Never read");
|
||||
expect(state.systemConfig.system_prompt).toContain(".project_config.toml");
|
||||
expect(state.systemConfig.system_prompt).toContain("agent_state/.vault.toml");
|
||||
expect(state.systemConfig.system_prompt).toContain("CLI-only");
|
||||
expect(state.systemConfig.system_prompt).toContain("penguin config");
|
||||
expect(state.systemConfig.system_prompt).not.toContain(".credentials.toml");
|
||||
expect(state.systemConfig.system_prompt.indexOf(AGENTS_MD_PLACEHOLDER)).toBeLessThan(
|
||||
state.systemConfig.system_prompt.indexOf("# Environment"),
|
||||
);
|
||||
// The # Vault and # Skills body sections plus their placeholders: the default template
|
||||
// places them after [/developer_instructions] and before # Environment, in the order
|
||||
// Vault -> Skills (the statement text is part of the template body, kept even with no
|
||||
// keys/skills).
|
||||
const tpl = state.systemConfig.system_prompt;
|
||||
expect(tpl).toContain("# Vault");
|
||||
expect(tpl).toContain(VAULT_KEYS_PLACEHOLDER);
|
||||
expect(tpl).toContain("# Skills");
|
||||
expect(tpl).toContain(SKILL_METADATA_PLACEHOLDER);
|
||||
expect(tpl).toContain("[use_skills]");
|
||||
expect(tpl.indexOf("[/developer_instructions]")).toBeLessThan(tpl.indexOf("# Vault"));
|
||||
expect(tpl.indexOf("# Vault")).toBeLessThan(tpl.indexOf(VAULT_KEYS_PLACEHOLDER));
|
||||
expect(tpl.indexOf(VAULT_KEYS_PLACEHOLDER)).toBeLessThan(tpl.indexOf("# Skills"));
|
||||
expect(tpl.indexOf("# Skills")).toBeLessThan(tpl.indexOf(SKILL_METADATA_PLACEHOLDER));
|
||||
expect(tpl.indexOf(SKILL_METADATA_PLACEHOLDER)).toBeLessThan(tpl.indexOf("# Environment"));
|
||||
expect(state.systemConfig.model?.max_tokens).toBe(32000);
|
||||
expect(state.systemConfig.model?.thinking_level).toBe("medium");
|
||||
expect(state.systemConfig.model?.timeoutMs).toBe(120000);
|
||||
expect(state.systemConfig.tools?.mcpServers).toEqual([]);
|
||||
expect(Object.hasOwn(state.systemConfig, "description")).toBe(false);
|
||||
expect(Object.hasOwn(state.systemConfig, "subagents")).toBe(false);
|
||||
},
|
||||
);
|
||||
|
||||
it("loads an existing agent directory and returns the same system prompt", async () => {
|
||||
const first = await loadOrInitAgentState();
|
||||
@@ -500,6 +513,7 @@ describe("assembleSystemPrompt", () => {
|
||||
`pdir=${PROJECT_DIR_PLACEHOLDER}`,
|
||||
`platform=${PLATFORM_PLACEHOLDER}`,
|
||||
`os=${OS_VERSION_PLACEHOLDER}`,
|
||||
`shell=${SHELL_PLACEHOLDER}`,
|
||||
`date=${DATE_PLACEHOLDER}`,
|
||||
"middle",
|
||||
AGENTS_MD_PLACEHOLDER,
|
||||
@@ -518,6 +532,7 @@ describe("assembleSystemPrompt", () => {
|
||||
modelId: "deepseek-v4-pro",
|
||||
platform: "darwin",
|
||||
osVersion: "Darwin 25.0.0",
|
||||
shell: "zsh",
|
||||
date: "2026-06-30",
|
||||
});
|
||||
expect(prompt).toBe(
|
||||
@@ -529,6 +544,7 @@ describe("assembleSystemPrompt", () => {
|
||||
"pdir=/tmp/proj",
|
||||
"platform=darwin",
|
||||
"os=Darwin 25.0.0",
|
||||
"shell=zsh",
|
||||
"date=2026-06-30",
|
||||
"middle",
|
||||
"# Agent Rules\nFollow local rules.",
|
||||
@@ -572,6 +588,7 @@ describe("assembleSystemPrompt", () => {
|
||||
modelId: "deepseek-v4-pro",
|
||||
platform: "darwin",
|
||||
osVersion: "Darwin 25.0.0",
|
||||
shell: "zsh",
|
||||
date: "2026-06-30",
|
||||
});
|
||||
expect(prompt).toBe("base prompt");
|
||||
@@ -657,9 +674,11 @@ describe("assembleSystemPrompt", () => {
|
||||
expect(prompt).toContain("Model ID: gpt-5.5");
|
||||
expect(prompt).toContain("Platform:");
|
||||
expect(prompt).toContain("OS Version:");
|
||||
expect(prompt).toContain("Shell:");
|
||||
expect(prompt).toContain("Date: 2026-06-30");
|
||||
expect(prompt.indexOf("Platform:")).toBeLessThan(prompt.indexOf("OS Version:"));
|
||||
expect(prompt.indexOf("OS Version:")).toBeLessThan(prompt.indexOf("Date:"));
|
||||
expect(prompt.indexOf("OS Version:")).toBeLessThan(prompt.indexOf("Shell:"));
|
||||
expect(prompt.indexOf("Shell:")).toBeLessThan(prompt.indexOf("Date:"));
|
||||
expect(prompt.indexOf("Date:")).toBeLessThan(prompt.indexOf("App Data Dir:"));
|
||||
expect(prompt.indexOf("App Data Dir:")).toBeLessThan(prompt.indexOf("Agent ID:"));
|
||||
expect(prompt.indexOf("Agent ID:")).toBeLessThan(prompt.indexOf("CWD:"));
|
||||
@@ -667,6 +686,114 @@ describe("assembleSystemPrompt", () => {
|
||||
expect(prompt.indexOf("Provider:")).toBeLessThan(prompt.indexOf("Model ID:"));
|
||||
expect(prompt.indexOf("Model ID:")).toBeLessThan(prompt.indexOf("Session ID:"));
|
||||
});
|
||||
|
||||
// The win32 Shell-line fallback for pre-{{SHELL}} templates (system_config.yaml is baked at
|
||||
// Agent creation and never auto-upgraded). Removable together with `withShellLineFallback`
|
||||
// once pre-{{SHELL}} Agent configs are no longer expected in the wild.
|
||||
describe("Shell line fallback for templates without {{SHELL}}", () => {
|
||||
const stateWithPrompt = (system_prompt: string) => ({
|
||||
root: tmpRoot,
|
||||
projectId: DEFAULT_PROJECT_ID,
|
||||
agentId: DEFAULT_AGENT_ID,
|
||||
stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID),
|
||||
systemConfig: { system_prompt },
|
||||
agentsMd: "",
|
||||
});
|
||||
const envFor = (platform: string) => ({
|
||||
sessionId: "session-1",
|
||||
cwd: "C:\\ws",
|
||||
agentId: "agent-x",
|
||||
projectDir: "C:\\proj",
|
||||
provider: "deepseek",
|
||||
modelId: "deepseek-v4-pro",
|
||||
platform,
|
||||
osVersion: "Windows 11 Pro 10.0.26100",
|
||||
shell: "pwsh",
|
||||
date: "2026-07-27",
|
||||
});
|
||||
// A pre-{{SHELL}} default-template Environment section (Platform/OS Version/Date, no Shell).
|
||||
const preShellTemplate = [
|
||||
"intro",
|
||||
"# Environment",
|
||||
`- Platform: ${PLATFORM_PLACEHOLDER}`,
|
||||
`- OS Version: ${OS_VERSION_PLACEHOLDER}`,
|
||||
`- Date: ${DATE_PLACEHOLDER}`,
|
||||
"",
|
||||
"# Tail section",
|
||||
"tail",
|
||||
].join("\n");
|
||||
|
||||
it("injects the line exactly once into the Environment section on win32", () => {
|
||||
const prompt = assembleSystemPrompt(stateWithPrompt(preShellTemplate), envFor("win32"));
|
||||
expect(prompt).toBe(
|
||||
[
|
||||
"intro",
|
||||
"# Environment",
|
||||
"- Shell: pwsh",
|
||||
"- Platform: win32",
|
||||
"- OS Version: Windows 11 Pro 10.0.26100",
|
||||
"- Date: 2026-07-27",
|
||||
"",
|
||||
"# Tail section",
|
||||
"tail",
|
||||
].join("\n"),
|
||||
);
|
||||
expect(prompt.split("- Shell: pwsh").length - 1).toBe(1);
|
||||
});
|
||||
|
||||
it("keeps POSIX output byte-identical (no injected line)", () => {
|
||||
for (const platform of ["linux", "darwin"]) {
|
||||
const prompt = assembleSystemPrompt(stateWithPrompt(preShellTemplate), {
|
||||
...envFor(platform),
|
||||
shell: "bash",
|
||||
osVersion: "Linux 6.1.0",
|
||||
});
|
||||
expect(prompt).toBe(
|
||||
[
|
||||
"intro",
|
||||
"# Environment",
|
||||
`- Platform: ${platform}`,
|
||||
"- OS Version: Linux 6.1.0",
|
||||
"- Date: 2026-07-27",
|
||||
"",
|
||||
"# Tail section",
|
||||
"tail",
|
||||
].join("\n"),
|
||||
);
|
||||
expect(prompt).not.toContain("Shell:");
|
||||
}
|
||||
});
|
||||
|
||||
it("does not duplicate the line when the template has {{SHELL}}", () => {
|
||||
const template = [
|
||||
"# Environment",
|
||||
`- Platform: ${PLATFORM_PLACEHOLDER}`,
|
||||
`- Shell: ${SHELL_PLACEHOLDER}`,
|
||||
].join("\n");
|
||||
const prompt = assembleSystemPrompt(stateWithPrompt(template), envFor("win32"));
|
||||
expect(prompt).toBe(["# Environment", "- Platform: win32", "- Shell: pwsh"].join("\n"));
|
||||
expect(prompt.split("- Shell:").length - 1).toBe(1);
|
||||
});
|
||||
|
||||
it("does not duplicate a hardcoded line and appends a minimal line without an Environment section", () => {
|
||||
// A custom template that hardcodes the exact line: left untouched (idempotent).
|
||||
const hardcoded = assembleSystemPrompt(
|
||||
stateWithPrompt("base prompt\n- Shell: pwsh"),
|
||||
envFor("win32"),
|
||||
);
|
||||
expect(hardcoded).toBe("base prompt\n- Shell: pwsh");
|
||||
// A hardcoded line with a different value is a deliberate template choice:
|
||||
// never add a second, contradicting Shell line.
|
||||
const pinned = assembleSystemPrompt(
|
||||
stateWithPrompt("base prompt\n- Shell: bash"),
|
||||
envFor("win32"),
|
||||
);
|
||||
expect(pinned).toBe("base prompt\n- Shell: bash");
|
||||
// No Environment section at all: the line is appended at the end.
|
||||
const appended = assembleSystemPrompt(stateWithPrompt("base prompt"), envFor("win32"));
|
||||
expect(appended).toBe("base prompt\n- Shell: pwsh");
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("resetSystemConfigToDefaults", () => {
|
||||
@@ -1174,28 +1301,35 @@ describe("single hidden config file (.project_config.toml, credentials inlined)"
|
||||
// The sole config file is hidden (not shown by ls by default) and has 0600 permission
|
||||
// (owner read/write only).
|
||||
expect(path.basename(file)).toBe(".project_config.toml");
|
||||
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
|
||||
// POSIX-only: Windows has no owner-only mode bits (chmod maps to the read-only attribute).
|
||||
if (process.platform !== "win32") {
|
||||
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
|
||||
}
|
||||
expect(await fs.readFile(file, "utf8")).toContain("sk-split-1");
|
||||
// The old two-file layout is no longer produced.
|
||||
expect(await exists(path.join(tmpRoot, DEFAULT_PROJECT_ID, "project_config.toml"))).toBe(false);
|
||||
expect(await exists(path.join(tmpRoot, DEFAULT_PROJECT_ID, ".credentials.toml"))).toBe(false);
|
||||
});
|
||||
|
||||
it("chmod converges an existing file back to 0600 on save", async () => {
|
||||
await addModel(tmpRoot, DEFAULT_PROJECT_ID, {
|
||||
provider: "custom",
|
||||
model_id: "m-perm",
|
||||
api_key: "sk-1",
|
||||
});
|
||||
const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID);
|
||||
await fs.chmod(file, 0o644);
|
||||
await addModel(tmpRoot, DEFAULT_PROJECT_ID, {
|
||||
provider: "custom",
|
||||
model_id: "m-perm",
|
||||
api_key: "sk-2",
|
||||
});
|
||||
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
|
||||
});
|
||||
// POSIX-only: Windows has no owner-only mode bits to converge.
|
||||
it.skipIf(process.platform === "win32")(
|
||||
"chmod converges an existing file back to 0600 on save",
|
||||
async () => {
|
||||
await addModel(tmpRoot, DEFAULT_PROJECT_ID, {
|
||||
provider: "custom",
|
||||
model_id: "m-perm",
|
||||
api_key: "sk-1",
|
||||
});
|
||||
const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID);
|
||||
await fs.chmod(file, 0o644);
|
||||
await addModel(tmpRoot, DEFAULT_PROJECT_ID, {
|
||||
provider: "custom",
|
||||
model_id: "m-perm",
|
||||
api_key: "sk-2",
|
||||
});
|
||||
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
|
||||
},
|
||||
);
|
||||
|
||||
it("writes provider and model_id as separate fields; refs are TOML inline tables", async () => {
|
||||
await addModel(
|
||||
@@ -1238,7 +1372,10 @@ describe("agent vault (agent_state/.vault.toml)", () => {
|
||||
// with 0600 permission (owner read/write only).
|
||||
const file = agentVaultPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID);
|
||||
expect(path.basename(file)).toBe(".vault.toml");
|
||||
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
|
||||
// POSIX-only: Windows has no owner-only mode bits (chmod maps to the read-only attribute).
|
||||
if (process.platform !== "win32") {
|
||||
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
|
||||
}
|
||||
const raw = await fs.readFile(file, "utf8");
|
||||
expect(raw).toContain("sk-secret-2");
|
||||
// The Project config no longer carries the vault.
|
||||
|
||||
Reference in New Issue
Block a user