feat(core,tooling): Windows support — shell selection, install.ps1, win-x64 release package, Windows CI (#79)

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Yaowei Zheng
2026-07-27 22:17:35 +08:00
committed by GitHub
parent c23f0bc2f0
commit c369e089a7
46 changed files with 1315 additions and 217 deletions
@@ -126,6 +126,7 @@ describe("App Data Dir / Agent ID placeholders", () => {
modelId: "deepseek-v4-pro",
platform: "linux",
osVersion: "test",
shell: "bash",
date: "2026-07-08",
});
expect(prompt).toContain("Agent ID: env_agent");
+17 -9
View File
@@ -132,8 +132,10 @@ describe("ContextEngine ReAct loop (mock LLM, approve callback)", () => {
});
afterEach(async () => {
await rm(workspace, { recursive: true, force: true });
await rm(traces, { recursive: true, force: true });
// Retries (here and in the other cleanups below): on Windows a just-killed process tree
// releases its cwd locks asynchronously, so an immediate rm can hit EBUSY.
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
await rm(traces, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
});
it("approves a tool call, writes the file, returns the final answer, traces it", async () => {
@@ -733,7 +735,7 @@ describe("ContextEngine async/incremental tool calls (overlapping execution)", (
workspace = await mkdtemp(join(tmpdir(), "penguin-ws2-"));
});
afterEach(async () => {
await rm(workspace, { recursive: true, force: true });
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
});
it("emits both tool calls in one round; second is approved while the first executes; outputs come back in completion order", async () => {
@@ -809,7 +811,13 @@ describe("ContextEngine async/incremental tool calls (overlapping execution)", (
// command has not finished yet) -- i.e., execution does not block the next approval.
expect(approvedAt["t2"]!).toBeLessThan(firstCompleteAt["t1"] ?? Infinity);
// The fast b.txt finishes first, the slow a.txt finishes later (outputs in completion order).
expect(firstCompleteAt["t2"]!).toBeLessThan(firstCompleteAt["t1"]!);
// POSIX only: on Windows a cold Git-Bash spawn costs 1-2s, which can swamp the 400ms sleep
// delta that makes t1 "the slow one" — CI has seen the two complete within 5ms — so the
// relative completion order is not controllable there. The overlap assertion above and the
// file contents below still run on Windows.
if (process.platform !== "win32") {
expect(firstCompleteAt["t2"]!).toBeLessThan(firstCompleteAt["t1"]!);
}
expect(await readFile(join(workspace, "a.txt"), "utf8")).toBe("one");
expect(await readFile(join(workspace, "b.txt"), "utf8")).toBe("two");
@@ -895,7 +903,7 @@ describe("ContextEngine tool execution resilience", () => {
workspace = await mkdtemp(join(tmpdir(), "penguin-ws4-"));
});
afterEach(async () => {
await rm(workspace, { recursive: true, force: true });
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
});
it("feeds a failed tool output back and keeps tool_use/result paired (Environment converges errors, never throws)", async () => {
@@ -1105,7 +1113,7 @@ describe("ContextEngine abort during execution", () => {
workspace = await mkdtemp(join(tmpdir(), "penguin-ws3-"));
});
afterEach(async () => {
await rm(workspace, { recursive: true, force: true });
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
});
it("aborting a long-running tool ends the turn, emits abort, and carries tool results over (model output completed)", async () => {
@@ -1255,7 +1263,7 @@ describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => {
workspace = await mkdtemp(join(tmpdir(), "penguin-ws4-"));
});
afterEach(async () => {
await rm(workspace, { recursive: true, force: true });
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
});
it("auto-retries on LLM timeout: original input + [turn_retried] carrying partial products", async () => {
@@ -1792,8 +1800,8 @@ describe("ContextEngine mid-run steering ([user_steering])", () => {
});
afterEach(async () => {
await rm(workspace, { recursive: true, force: true });
await rm(traces, { recursive: true, force: true });
await rm(workspace, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
await rm(traces, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
});
/** Fake environment: streams a delta then closes with a fixed complete output (no real shell). */
+16 -6
View File
@@ -67,7 +67,9 @@ beforeEach(async () => {
afterEach(async () => {
if (originalHome === undefined) delete process.env.HOME;
else process.env.HOME = originalHome;
await rm(tmp, { recursive: true, force: true });
// Retries: on Windows a just-killed process tree releases its cwd/file locks asynchronously,
// so an immediate recursive rm can hit EBUSY; fs.rm retries those with a linear backoff.
await rm(tmp, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
});
describe("Environment.listTools", () => {
@@ -485,9 +487,12 @@ describe("Environment.executeTool — relaxed tool contract", () => {
describe("Environment.executeTool — timeoutMs (PRN-013)", () => {
it("fails a tool exceeding timeoutMs, keeps prior output, and streams the timeout reason", async () => {
// The timeout must stay below MIN_YIELD_MS (250): for larger values exec_command yields to
// background (with a process_id) before the Environment timeout can ever fire.
const timeoutMs = 200;
const env = new Environment({
workspaceDir: tmp,
toolConfig: makeToolConfig(execTool({ timeoutMs: 200 })),
toolConfig: makeToolConfig(execTool({ timeoutMs })),
});
const startedAt = Date.now();
@@ -513,8 +518,13 @@ describe("Environment.executeTool — timeoutMs (PRN-013)", () => {
};
expect(last.type).toBe("tool_call_output");
expect(last.stop_reason).toBe("failed");
expect(last.output).toContain("begin");
expect(last.output).toContain("[tool timeout: exceeded 200ms]");
// Kept-prior-output is asserted only where the shell can win the race: a Git-Bash login
// shell on Windows needs several hundred ms to start, so nothing is printed before a
// sub-250ms timeout there — the timeout mechanics above are still fully exercised.
if (process.platform !== "win32") {
expect(last.output).toContain("begin");
}
expect(last.output).toContain(`[tool timeout: exceeded ${timeoutMs}ms]`);
// The timeout marker is also produced via streaming: concatenating the streamed deltas ==
// the complete content.
const streamed = messages
@@ -735,14 +745,14 @@ describe("Environment.executeTool — robustness", () => {
describe("Environment.toolPermission", () => {
it("returns the configured permission for a known tool", () => {
const env = new Environment({
workspaceDir: "/tmp",
workspaceDir: tmpdir(),
toolConfig: makeToolConfig(execTool({ permission: "rw" })),
});
expect(env.toolPermission("exec_command")).toBe("rw");
});
it("returns undefined for an unknown tool", () => {
const env = new Environment({ workspaceDir: "/tmp", toolConfig: makeToolConfig() });
const env = new Environment({ workspaceDir: tmpdir(), toolConfig: makeToolConfig() });
expect(env.toolPermission("nope")).toBeUndefined();
});
});
+3 -1
View File
@@ -91,7 +91,9 @@ beforeEach(async () => {
afterEach(async () => {
env.dispose();
await rm(tmp, { recursive: true, force: true });
// Retries: on Windows a just-killed process tree releases its cwd/file locks asynchronously,
// so an immediate recursive rm can hit EBUSY; fs.rm retries those with a linear backoff.
await rm(tmp, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
});
describe("exec_command — long-running command sessions", () => {
+13 -9
View File
@@ -514,15 +514,19 @@ describe("edit_file — review follow-ups", () => {
describe("write_file — review follow-ups", () => {
const tool = () => createWriteFileTool(def(WRITE_FILE_NAME, "rw"));
it("preserves the permission bits of an overwritten file (atomic rename)", async () => {
const file = path.join(tmp, "mode.txt");
await writeFile(file, "old");
await chmod(file, 0o600);
const { result } = await run(tool(), { file_path: "mode.txt", content: "new" }, tmp);
expect(result?.stopReason).toBeUndefined();
expect((await stat(file)).mode & 0o777).toBe(0o600);
expect(await readFile(file, "utf8")).toBe("new");
});
// POSIX-only: Windows has no owner-only mode bits to preserve (chmod maps to the read-only attribute).
it.skipIf(process.platform === "win32")(
"preserves the permission bits of an overwritten file (atomic rename)",
async () => {
const file = path.join(tmp, "mode.txt");
await writeFile(file, "old");
await chmod(file, 0o600);
const { result } = await run(tool(), { file_path: "mode.txt", content: "new" }, tmp);
expect(result?.stopReason).toBeUndefined();
expect((await stat(file)).mode & 0o777).toBe(0o600);
expect(await readFile(file, "utf8")).toBe("new");
},
);
it("writes atomically: no temp files are left behind", async () => {
await run(tool(), { file_path: "fresh.txt", content: "x" }, tmp);
+112
View File
@@ -0,0 +1,112 @@
/**
* Unit tests for the command-session shell resolver (pure function; platform, env and the
* PATH probe are injected — no real shells are spawned here).
*/
import { describe, expect, it } from "vitest";
import { resolveShell } from "../src/environment/tools/command/shell.js";
const POWERSHELL_ARGS = ["-NoLogo", "-NoProfile", "-Command"];
/** A whichAll stub resolving only the given names (value = returned PATH matches). */
function which(table: Record<string, string[]>): (cmd: string) => string[] {
return (cmd) => table[cmd] ?? [];
}
describe("resolveShell — POSIX", () => {
it("uses bash -lc on linux without probing (today's behavior, unchanged)", () => {
let probed = false;
const shell = resolveShell({
platform: "linux",
env: {},
whichAll: () => {
probed = true;
return [];
},
});
expect(shell).toEqual({ command: "bash", args: ["-lc"], name: "bash" });
expect(probed).toBe(false);
});
it("uses bash -lc on darwin", () => {
const shell = resolveShell({ platform: "darwin", env: {} });
expect(shell).toEqual({ command: "bash", args: ["-lc"], name: "bash" });
});
});
describe("resolveShell — win32 probing", () => {
it("prefers bash on PATH (Git for Windows)", () => {
const shell = resolveShell({
platform: "win32",
env: {},
whichAll: which({
bash: ["C:\\Program Files\\Git\\bin\\bash.exe"],
pwsh: ["C:\\Program Files\\PowerShell\\7\\pwsh.exe"],
}),
});
expect(shell).toEqual({ command: "bash", args: ["-lc"], name: "bash" });
});
it("skips the WSL launcher bash under the system root and falls through to pwsh", () => {
const shell = resolveShell({
platform: "win32",
env: { SystemRoot: "C:\\WINDOWS" },
whichAll: which({
bash: ["C:\\Windows\\System32\\bash.exe"],
pwsh: ["C:\\Program Files\\PowerShell\\7\\pwsh.exe"],
}),
});
expect(shell).toEqual({ command: "pwsh", args: POWERSHELL_ARGS, name: "pwsh" });
});
it("falls back to pwsh when bash is absent", () => {
const shell = resolveShell({
platform: "win32",
env: {},
whichAll: which({ pwsh: ["C:\\Program Files\\PowerShell\\7\\pwsh.exe"] }),
});
expect(shell).toEqual({ command: "pwsh", args: POWERSHELL_ARGS, name: "pwsh" });
});
it("falls back to powershell when neither bash nor pwsh resolve", () => {
const shell = resolveShell({ platform: "win32", env: {}, whichAll: which({}) });
expect(shell).toEqual({ command: "powershell", args: POWERSHELL_ARGS, name: "powershell" });
});
});
describe("resolveShell — PENGUIN_SHELL override", () => {
it("wins on every platform and keeps POSIX-style args for a POSIX shell path", () => {
const shell = resolveShell({
platform: "linux",
env: { PENGUIN_SHELL: "/usr/bin/zsh" },
});
expect(shell).toEqual({ command: "/usr/bin/zsh", args: ["-lc"], name: "zsh" });
});
it("uses PowerShell-style args when the basename is pwsh (case/extension-insensitive)", () => {
const shell = resolveShell({
platform: "win32",
env: { PENGUIN_SHELL: "C:\\Program Files\\PowerShell\\7\\pwsh.EXE" },
whichAll: which({ bash: ["C:\\Program Files\\Git\\bin\\bash.exe"] }),
});
expect(shell).toEqual({
command: "C:\\Program Files\\PowerShell\\7\\pwsh.EXE",
args: POWERSHELL_ARGS,
name: "pwsh",
});
});
it("uses PowerShell-style args for a bare powershell name", () => {
const shell = resolveShell({ platform: "win32", env: { PENGUIN_SHELL: "powershell" } });
expect(shell).toEqual({ command: "powershell", args: POWERSHELL_ARGS, name: "powershell" });
});
it("uses cmd-style args when the basename is cmd", () => {
const shell = resolveShell({ platform: "win32", env: { PENGUIN_SHELL: "cmd" } });
expect(shell).toEqual({ command: "cmd", args: ["/d", "/s", "/c"], name: "cmd" });
});
it("ignores a blank PENGUIN_SHELL", () => {
const shell = resolveShell({ platform: "linux", env: { PENGUIN_SHELL: " " } });
expect(shell).toEqual({ command: "bash", args: ["-lc"], name: "bash" });
});
});
+234 -97
View File
@@ -16,6 +16,7 @@ import {
PLATFORM_PLACEHOLDER,
PROJECT_DIR_PLACEHOLDER,
SESSION_ID_PLACEHOLDER,
SHELL_PLACEHOLDER,
addModel,
setVisionModel,
agentsMdPath,
@@ -64,7 +65,10 @@ afterEach(async () => {
} else {
process.env.PENGUIN_HOME = prevHome;
}
await fs.rm(tmpRoot, { recursive: true, force: true });
// Retries: when a test times out, vitest runs this cleanup while the test's un-cancelled
// init may still be writing files, so an immediate recursive rm can hit ENOTEMPTY on
// Windows (fs.rm retries ENOTEMPTY/EBUSY/EPERM); a no-op when removal succeeds first try.
await fs.rm(tmpRoot, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 });
});
async function exists(p: string): Promise<boolean> {
@@ -83,87 +87,96 @@ describe("paths / resolveRoot", () => {
});
describe("loadOrInitAgentState", () => {
it("initializes an empty agent directory with the full state layout", async () => {
const state = await loadOrInitAgentState();
expect(state.root).toBe(tmpRoot);
expect(state.projectId).toBe(DEFAULT_PROJECT_ID);
expect(state.agentId).toBe(DEFAULT_AGENT_ID);
// Timeout: initialization writes the full layout — 15 library skills plus the example
// benchmark, dozens of small files — and this first init test also pays the cold-I/O cost
// (first-touch reads of the skills package, Defender scans) on Windows runners, where a
// slow-disk moment has pushed it past the 5s default. Purely a failure deadline: passing
// runs stay as fast as before on every platform.
it(
"initializes an empty agent directory with the full state layout",
{ timeout: 20_000 },
async () => {
const state = await loadOrInitAgentState();
expect(state.root).toBe(tmpRoot);
expect(state.projectId).toBe(DEFAULT_PROJECT_ID);
expect(state.agentId).toBe(DEFAULT_AGENT_ID);
const root = tmpRoot;
expect(await exists(systemConfigPath(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
expect(await exists(agentsMdPath(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
expect(await exists(toolsDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
expect(await exists(memoryDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
expect(await exists(skillsDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
// The scratchpad/ directory alongside agent_state (model temp files get a subdirectory per Session id).
expect(await exists(scratchpadDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
const root = tmpRoot;
expect(await exists(systemConfigPath(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
expect(await exists(agentsMdPath(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
expect(await exists(toolsDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
expect(await exists(memoryDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
expect(await exists(skillsDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
// The scratchpad/ directory alongside agent_state (model temp files get a subdirectory per Session id).
expect(await exists(scratchpadDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID))).toBe(true);
expect(state.stateDir).toBe(agentStateDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID));
expect(state.stateDir).toBe(agentStateDir(root, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID));
// The default system Prompt states the Agent's identity, without repeating tool details
// already in the tool schema (Suggested workflows only points to the run_subagent
// delegation entry point).
expect(state.systemConfig.system_prompt).toContain("PenguinHarness");
expect(state.systemConfig.system_prompt).not.toContain("exec_command");
// Suggested workflows absorbs Subagent delegation and task conventions (self-reported
// identity as a soft convention, parallelism, file exchange).
expect(state.systemConfig.system_prompt).toContain("# Suggested workflows");
expect(state.systemConfig.system_prompt).toContain("run_subagent");
expect(state.systemConfig.system_prompt).toContain("Caller agent");
// The default AGENTS.md is empty: it carries no preset guidance.
expect(state.agentsMd).toBe("");
expect(state.systemConfig.system_prompt).toContain(AGENTS_MD_PLACEHOLDER);
expect(state.systemConfig.system_prompt).toContain(SESSION_ID_PLACEHOLDER);
expect(state.systemConfig.system_prompt).toContain(CWD_PLACEHOLDER);
expect(state.systemConfig.system_prompt).toContain(PLATFORM_PLACEHOLDER);
expect(state.systemConfig.system_prompt).toContain(OS_VERSION_PLACEHOLDER);
expect(state.systemConfig.system_prompt).toContain(DATE_PLACEHOLDER);
// AGENTS.md and the Environment injection sit at the end of the template, with AGENTS.md
// before Environment; the [developer_instructions] wrapper text is written directly into
// the template (the Prompt is transparent about the config).
expect(state.systemConfig.system_prompt).toContain("[developer_instructions]");
expect(state.systemConfig.system_prompt).toContain("[/developer_instructions]");
// The default template explains the semantics of system-synthesized markers to the model,
// and recommends preferring tool use.
expect(state.systemConfig.system_prompt).toContain("[turn_aborted]");
expect(state.systemConfig.system_prompt).toContain("[turn_retried]");
expect(state.systemConfig.system_prompt).toContain("[context_summary]");
expect(state.systemConfig.system_prompt).toContain("[user_steering]");
expect(state.systemConfig.system_prompt).toContain("# Tool use");
// Privacy hardening: explicitly forbids reading .project_config.toml (the sole config file,
// which holds API keys) and each Agent's .vault.toml, and states that config can only be
// changed via the CLI (penguin config ...).
expect(state.systemConfig.system_prompt).toContain("Never read");
expect(state.systemConfig.system_prompt).toContain(".project_config.toml");
expect(state.systemConfig.system_prompt).toContain("agent_state/.vault.toml");
expect(state.systemConfig.system_prompt).toContain("CLI-only");
expect(state.systemConfig.system_prompt).toContain("penguin config");
expect(state.systemConfig.system_prompt).not.toContain(".credentials.toml");
expect(state.systemConfig.system_prompt.indexOf(AGENTS_MD_PLACEHOLDER)).toBeLessThan(
state.systemConfig.system_prompt.indexOf("# Environment"),
);
// The # Vault and # Skills body sections plus their placeholders: the default template
// places them after [/developer_instructions] and before # Environment, in the order
// Vault -> Skills (the statement text is part of the template body, kept even with no
// keys/skills).
const tpl = state.systemConfig.system_prompt;
expect(tpl).toContain("# Vault");
expect(tpl).toContain(VAULT_KEYS_PLACEHOLDER);
expect(tpl).toContain("# Skills");
expect(tpl).toContain(SKILL_METADATA_PLACEHOLDER);
expect(tpl).toContain("[use_skills]");
expect(tpl.indexOf("[/developer_instructions]")).toBeLessThan(tpl.indexOf("# Vault"));
expect(tpl.indexOf("# Vault")).toBeLessThan(tpl.indexOf(VAULT_KEYS_PLACEHOLDER));
expect(tpl.indexOf(VAULT_KEYS_PLACEHOLDER)).toBeLessThan(tpl.indexOf("# Skills"));
expect(tpl.indexOf("# Skills")).toBeLessThan(tpl.indexOf(SKILL_METADATA_PLACEHOLDER));
expect(tpl.indexOf(SKILL_METADATA_PLACEHOLDER)).toBeLessThan(tpl.indexOf("# Environment"));
expect(state.systemConfig.model?.max_tokens).toBe(32000);
expect(state.systemConfig.model?.thinking_level).toBe("medium");
expect(state.systemConfig.model?.timeoutMs).toBe(120000);
expect(state.systemConfig.tools?.mcpServers).toEqual([]);
expect(Object.hasOwn(state.systemConfig, "description")).toBe(false);
expect(Object.hasOwn(state.systemConfig, "subagents")).toBe(false);
});
// The default system Prompt states the Agent's identity, without repeating tool details
// already in the tool schema (Suggested workflows only points to the run_subagent
// delegation entry point).
expect(state.systemConfig.system_prompt).toContain("PenguinHarness");
expect(state.systemConfig.system_prompt).not.toContain("exec_command");
// Suggested workflows absorbs Subagent delegation and task conventions (self-reported
// identity as a soft convention, parallelism, file exchange).
expect(state.systemConfig.system_prompt).toContain("# Suggested workflows");
expect(state.systemConfig.system_prompt).toContain("run_subagent");
expect(state.systemConfig.system_prompt).toContain("Caller agent");
// The default AGENTS.md is empty: it carries no preset guidance.
expect(state.agentsMd).toBe("");
expect(state.systemConfig.system_prompt).toContain(AGENTS_MD_PLACEHOLDER);
expect(state.systemConfig.system_prompt).toContain(SESSION_ID_PLACEHOLDER);
expect(state.systemConfig.system_prompt).toContain(CWD_PLACEHOLDER);
expect(state.systemConfig.system_prompt).toContain(PLATFORM_PLACEHOLDER);
expect(state.systemConfig.system_prompt).toContain(OS_VERSION_PLACEHOLDER);
expect(state.systemConfig.system_prompt).toContain(DATE_PLACEHOLDER);
// AGENTS.md and the Environment injection sit at the end of the template, with AGENTS.md
// before Environment; the [developer_instructions] wrapper text is written directly into
// the template (the Prompt is transparent about the config).
expect(state.systemConfig.system_prompt).toContain("[developer_instructions]");
expect(state.systemConfig.system_prompt).toContain("[/developer_instructions]");
// The default template explains the semantics of system-synthesized markers to the model,
// and recommends preferring tool use.
expect(state.systemConfig.system_prompt).toContain("[turn_aborted]");
expect(state.systemConfig.system_prompt).toContain("[turn_retried]");
expect(state.systemConfig.system_prompt).toContain("[context_summary]");
expect(state.systemConfig.system_prompt).toContain("[user_steering]");
expect(state.systemConfig.system_prompt).toContain("# Tool use");
// Privacy hardening: explicitly forbids reading .project_config.toml (the sole config file,
// which holds API keys) and each Agent's .vault.toml, and states that config can only be
// changed via the CLI (penguin config ...).
expect(state.systemConfig.system_prompt).toContain("Never read");
expect(state.systemConfig.system_prompt).toContain(".project_config.toml");
expect(state.systemConfig.system_prompt).toContain("agent_state/.vault.toml");
expect(state.systemConfig.system_prompt).toContain("CLI-only");
expect(state.systemConfig.system_prompt).toContain("penguin config");
expect(state.systemConfig.system_prompt).not.toContain(".credentials.toml");
expect(state.systemConfig.system_prompt.indexOf(AGENTS_MD_PLACEHOLDER)).toBeLessThan(
state.systemConfig.system_prompt.indexOf("# Environment"),
);
// The # Vault and # Skills body sections plus their placeholders: the default template
// places them after [/developer_instructions] and before # Environment, in the order
// Vault -> Skills (the statement text is part of the template body, kept even with no
// keys/skills).
const tpl = state.systemConfig.system_prompt;
expect(tpl).toContain("# Vault");
expect(tpl).toContain(VAULT_KEYS_PLACEHOLDER);
expect(tpl).toContain("# Skills");
expect(tpl).toContain(SKILL_METADATA_PLACEHOLDER);
expect(tpl).toContain("[use_skills]");
expect(tpl.indexOf("[/developer_instructions]")).toBeLessThan(tpl.indexOf("# Vault"));
expect(tpl.indexOf("# Vault")).toBeLessThan(tpl.indexOf(VAULT_KEYS_PLACEHOLDER));
expect(tpl.indexOf(VAULT_KEYS_PLACEHOLDER)).toBeLessThan(tpl.indexOf("# Skills"));
expect(tpl.indexOf("# Skills")).toBeLessThan(tpl.indexOf(SKILL_METADATA_PLACEHOLDER));
expect(tpl.indexOf(SKILL_METADATA_PLACEHOLDER)).toBeLessThan(tpl.indexOf("# Environment"));
expect(state.systemConfig.model?.max_tokens).toBe(32000);
expect(state.systemConfig.model?.thinking_level).toBe("medium");
expect(state.systemConfig.model?.timeoutMs).toBe(120000);
expect(state.systemConfig.tools?.mcpServers).toEqual([]);
expect(Object.hasOwn(state.systemConfig, "description")).toBe(false);
expect(Object.hasOwn(state.systemConfig, "subagents")).toBe(false);
},
);
it("loads an existing agent directory and returns the same system prompt", async () => {
const first = await loadOrInitAgentState();
@@ -500,6 +513,7 @@ describe("assembleSystemPrompt", () => {
`pdir=${PROJECT_DIR_PLACEHOLDER}`,
`platform=${PLATFORM_PLACEHOLDER}`,
`os=${OS_VERSION_PLACEHOLDER}`,
`shell=${SHELL_PLACEHOLDER}`,
`date=${DATE_PLACEHOLDER}`,
"middle",
AGENTS_MD_PLACEHOLDER,
@@ -518,6 +532,7 @@ describe("assembleSystemPrompt", () => {
modelId: "deepseek-v4-pro",
platform: "darwin",
osVersion: "Darwin 25.0.0",
shell: "zsh",
date: "2026-06-30",
});
expect(prompt).toBe(
@@ -529,6 +544,7 @@ describe("assembleSystemPrompt", () => {
"pdir=/tmp/proj",
"platform=darwin",
"os=Darwin 25.0.0",
"shell=zsh",
"date=2026-06-30",
"middle",
"# Agent Rules\nFollow local rules.",
@@ -572,6 +588,7 @@ describe("assembleSystemPrompt", () => {
modelId: "deepseek-v4-pro",
platform: "darwin",
osVersion: "Darwin 25.0.0",
shell: "zsh",
date: "2026-06-30",
});
expect(prompt).toBe("base prompt");
@@ -657,9 +674,11 @@ describe("assembleSystemPrompt", () => {
expect(prompt).toContain("Model ID: gpt-5.5");
expect(prompt).toContain("Platform:");
expect(prompt).toContain("OS Version:");
expect(prompt).toContain("Shell:");
expect(prompt).toContain("Date: 2026-06-30");
expect(prompt.indexOf("Platform:")).toBeLessThan(prompt.indexOf("OS Version:"));
expect(prompt.indexOf("OS Version:")).toBeLessThan(prompt.indexOf("Date:"));
expect(prompt.indexOf("OS Version:")).toBeLessThan(prompt.indexOf("Shell:"));
expect(prompt.indexOf("Shell:")).toBeLessThan(prompt.indexOf("Date:"));
expect(prompt.indexOf("Date:")).toBeLessThan(prompt.indexOf("App Data Dir:"));
expect(prompt.indexOf("App Data Dir:")).toBeLessThan(prompt.indexOf("Agent ID:"));
expect(prompt.indexOf("Agent ID:")).toBeLessThan(prompt.indexOf("CWD:"));
@@ -667,6 +686,114 @@ describe("assembleSystemPrompt", () => {
expect(prompt.indexOf("Provider:")).toBeLessThan(prompt.indexOf("Model ID:"));
expect(prompt.indexOf("Model ID:")).toBeLessThan(prompt.indexOf("Session ID:"));
});
// The win32 Shell-line fallback for pre-{{SHELL}} templates (system_config.yaml is baked at
// Agent creation and never auto-upgraded). Removable together with `withShellLineFallback`
// once pre-{{SHELL}} Agent configs are no longer expected in the wild.
describe("Shell line fallback for templates without {{SHELL}}", () => {
const stateWithPrompt = (system_prompt: string) => ({
root: tmpRoot,
projectId: DEFAULT_PROJECT_ID,
agentId: DEFAULT_AGENT_ID,
stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID),
systemConfig: { system_prompt },
agentsMd: "",
});
const envFor = (platform: string) => ({
sessionId: "session-1",
cwd: "C:\\ws",
agentId: "agent-x",
projectDir: "C:\\proj",
provider: "deepseek",
modelId: "deepseek-v4-pro",
platform,
osVersion: "Windows 11 Pro 10.0.26100",
shell: "pwsh",
date: "2026-07-27",
});
// A pre-{{SHELL}} default-template Environment section (Platform/OS Version/Date, no Shell).
const preShellTemplate = [
"intro",
"# Environment",
`- Platform: ${PLATFORM_PLACEHOLDER}`,
`- OS Version: ${OS_VERSION_PLACEHOLDER}`,
`- Date: ${DATE_PLACEHOLDER}`,
"",
"# Tail section",
"tail",
].join("\n");
it("injects the line exactly once into the Environment section on win32", () => {
const prompt = assembleSystemPrompt(stateWithPrompt(preShellTemplate), envFor("win32"));
expect(prompt).toBe(
[
"intro",
"# Environment",
"- Shell: pwsh",
"- Platform: win32",
"- OS Version: Windows 11 Pro 10.0.26100",
"- Date: 2026-07-27",
"",
"# Tail section",
"tail",
].join("\n"),
);
expect(prompt.split("- Shell: pwsh").length - 1).toBe(1);
});
it("keeps POSIX output byte-identical (no injected line)", () => {
for (const platform of ["linux", "darwin"]) {
const prompt = assembleSystemPrompt(stateWithPrompt(preShellTemplate), {
...envFor(platform),
shell: "bash",
osVersion: "Linux 6.1.0",
});
expect(prompt).toBe(
[
"intro",
"# Environment",
`- Platform: ${platform}`,
"- OS Version: Linux 6.1.0",
"- Date: 2026-07-27",
"",
"# Tail section",
"tail",
].join("\n"),
);
expect(prompt).not.toContain("Shell:");
}
});
it("does not duplicate the line when the template has {{SHELL}}", () => {
const template = [
"# Environment",
`- Platform: ${PLATFORM_PLACEHOLDER}`,
`- Shell: ${SHELL_PLACEHOLDER}`,
].join("\n");
const prompt = assembleSystemPrompt(stateWithPrompt(template), envFor("win32"));
expect(prompt).toBe(["# Environment", "- Platform: win32", "- Shell: pwsh"].join("\n"));
expect(prompt.split("- Shell:").length - 1).toBe(1);
});
it("does not duplicate a hardcoded line and appends a minimal line without an Environment section", () => {
// A custom template that hardcodes the exact line: left untouched (idempotent).
const hardcoded = assembleSystemPrompt(
stateWithPrompt("base prompt\n- Shell: pwsh"),
envFor("win32"),
);
expect(hardcoded).toBe("base prompt\n- Shell: pwsh");
// A hardcoded line with a different value is a deliberate template choice:
// never add a second, contradicting Shell line.
const pinned = assembleSystemPrompt(
stateWithPrompt("base prompt\n- Shell: bash"),
envFor("win32"),
);
expect(pinned).toBe("base prompt\n- Shell: bash");
// No Environment section at all: the line is appended at the end.
const appended = assembleSystemPrompt(stateWithPrompt("base prompt"), envFor("win32"));
expect(appended).toBe("base prompt\n- Shell: pwsh");
});
});
});
describe("resetSystemConfigToDefaults", () => {
@@ -1174,28 +1301,35 @@ describe("single hidden config file (.project_config.toml, credentials inlined)"
// The sole config file is hidden (not shown by ls by default) and has 0600 permission
// (owner read/write only).
expect(path.basename(file)).toBe(".project_config.toml");
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
// POSIX-only: Windows has no owner-only mode bits (chmod maps to the read-only attribute).
if (process.platform !== "win32") {
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
}
expect(await fs.readFile(file, "utf8")).toContain("sk-split-1");
// The old two-file layout is no longer produced.
expect(await exists(path.join(tmpRoot, DEFAULT_PROJECT_ID, "project_config.toml"))).toBe(false);
expect(await exists(path.join(tmpRoot, DEFAULT_PROJECT_ID, ".credentials.toml"))).toBe(false);
});
it("chmod converges an existing file back to 0600 on save", async () => {
await addModel(tmpRoot, DEFAULT_PROJECT_ID, {
provider: "custom",
model_id: "m-perm",
api_key: "sk-1",
});
const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID);
await fs.chmod(file, 0o644);
await addModel(tmpRoot, DEFAULT_PROJECT_ID, {
provider: "custom",
model_id: "m-perm",
api_key: "sk-2",
});
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
});
// POSIX-only: Windows has no owner-only mode bits to converge.
it.skipIf(process.platform === "win32")(
"chmod converges an existing file back to 0600 on save",
async () => {
await addModel(tmpRoot, DEFAULT_PROJECT_ID, {
provider: "custom",
model_id: "m-perm",
api_key: "sk-1",
});
const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID);
await fs.chmod(file, 0o644);
await addModel(tmpRoot, DEFAULT_PROJECT_ID, {
provider: "custom",
model_id: "m-perm",
api_key: "sk-2",
});
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
},
);
it("writes provider and model_id as separate fields; refs are TOML inline tables", async () => {
await addModel(
@@ -1238,7 +1372,10 @@ describe("agent vault (agent_state/.vault.toml)", () => {
// with 0600 permission (owner read/write only).
const file = agentVaultPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID);
expect(path.basename(file)).toBe(".vault.toml");
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
// POSIX-only: Windows has no owner-only mode bits (chmod maps to the read-only attribute).
if (process.platform !== "win32") {
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
}
const raw = await fs.readFile(file, "utf8");
expect(raw).toContain("sk-secret-2");
// The Project config no longer carries the vault.