feat(core,cli,server,web): add goal mode — loop Tasks on one Session until an objective completes (#66)

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
rank-Yu
2026-07-27 23:43:24 +08:00
committed by GitHub
parent e1141ca010
commit 46463bee26
61 changed files with 3411 additions and 188 deletions
+23 -2
View File
@@ -4,8 +4,9 @@
* penguin chat [--model-id <id> --provider <group>] [--project-id <id>] [--agent-id <id>]
* [--workspace <path>] [--approve <allow-all|deny-all|read-only|always-ask>]
*
* Each line of input starts one conversation turn; `/compact` proactively compacts the
* context (reason=manual); `/exit` or `/quit` exits.
* Each line of input starts one conversation turn; `/goal[:<budget>] <objective>` runs
* goal mode (looping until the goal reaches a terminal state);
* `/compact` proactively compacts the context (reason=manual); `/exit` or `/quit` exits.
* Uses the current directory when no Workspace is specified. A model reference is always an
* explicit `(provider, model_id)` pair, so `--model-id` and `--provider` must be given
* together; giving neither uses the Project's default model.
@@ -30,6 +31,7 @@ import { createAgent, userText, VERSION } from "@prismshadow/penguin-core";
import type { ApprovalDecision, OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core";
import { StreamRenderer, dim, renderHistory, sessionMetaTools } from "../render.js";
import { runTask } from "../task-loop.js";
import { parseGoalCommand } from "../goal-command.js";
import { parseApprovalAnswer, resolveApprovalMode } from "../approval.js";
import { LineComposer, PasteFilter } from "../input.js";
import type { Messages } from "../i18n.js";
@@ -375,6 +377,25 @@ export function registerChatCommand(program: Command, t: Messages): void {
renderer.endCompact(Date.now() - startedAt);
}
if (!sawMessage) out.write(`${t.compactNothing()}\n`);
} else if (text === "/goal" || text.startsWith("/goal:") || text.startsWith("/goal ")) {
// Goal mode: one command drives the whole loop; Ctrl-C aborts the entire
// goal (a single signal spans every round), never just the current round.
const parsed = parseGoalCommand(text);
if (!parsed.ok) {
const message =
parsed.reason === "budget" ? t.goalBudgetInvalid(parsed.value) : t.goalUsage();
out.write(`${t.error(message)}\n`);
} else {
resumable = true;
await runTask(session, [userText(parsed.objective)], {
mode,
signal: taskAbort.signal,
renderer,
interactivePrompt,
t,
goal: { budget: parsed.budget, out },
});
}
} else {
resumable = true;
await runTask(session, [userText(text)], {
+49 -10
View File
@@ -4,18 +4,23 @@
* penguin run -m <msg> [--model-id <id> --provider <group>] [--workspace <path>]
* [--project-id <id>] [--agent-id <id>]
* [--approve <allow-all|deny-all|read-only|always-ask>]
* [--goal [budget]]
*
* Uses the current directory when Workspace is unspecified; uses the Project's default model
* when model is unspecified. A model reference is always an explicit `(provider, model_id)`
* pair, so `--model-id` and `--provider` must be given together — giving only one of them is
* an error, never a lookup. Defaults to interactive per-call approval; `--approve`
* selects the permission mode.
* `--goal` switches to goal mode: `-m` becomes the objective and the run loops until the
* goal reaches a terminal state (optional value = token budget, e.g. `--goal 500k`); only a
* completed goal exits 0.
* Docs: /docs/cli § "penguin run".
*/
import type { Command } from "commander";
import { createAgent, userText, VERSION } from "@prismshadow/penguin-core";
import { UNLIMITED_BUDGET, createAgent, userText, VERSION } from "@prismshadow/penguin-core";
import { StreamRenderer, sessionMetaTools } from "../render.js";
import { runTask } from "../task-loop.js";
import { parseTokenBudget } from "../goal-command.js";
import { denyActivePrompt, resolveApprovalMode } from "../approval.js";
import type { Messages } from "../i18n.js";
@@ -30,6 +35,7 @@ export function registerRunCommand(program: Command, t: Messages): void {
.option("--agent-id <id>", t.common.agentId)
.option("--workspace <path>", t.common.workspace)
.option("--approve <mode>", t.common.approve)
.option("--goal [budget]", t.run.goal)
.action(async (opts) => {
// The model reference is a pair: commander can only require each option on its own,
// so the "both or neither" rule is enforced here. Giving neither is the normal case
@@ -39,6 +45,24 @@ export function registerRunCommand(program: Command, t: Messages): void {
process.exitCode = 1;
return;
}
// --goal's optional value is the token budget (`--goal 500k`); a bare --goal means no
// budget. Validated before any Session is created, like the model-pair check above.
let goalBudget: number | null = null;
if (opts.goal !== undefined) {
goalBudget = opts.goal === true ? UNLIMITED_BUDGET : parseTokenBudget(String(opts.goal));
if (goalBudget === null) {
process.stderr.write(`${t.error(t.goalBudgetInvalid(String(opts.goal)))}\n`);
process.exitCode = 1;
return;
}
// The objective must be non-empty text (core throws on an empty one — turn the
// programming-level error into a friendly refusal before any Session exists).
if (String(opts.message).trim() === "") {
process.stderr.write(`${t.error(t.goalObjectiveEmpty())}\n`);
process.exitCode = 1;
return;
}
}
const mode = resolveApprovalMode(opts.approve, t);
const agent = await createAgent({
@@ -70,15 +94,30 @@ export function registerRunCommand(program: Command, t: Messages): void {
// The assembled tool schemas decide each tool's call-line preview path (see render.ts).
renderer.useToolSchemas(sessionMetaTools(session));
try {
const result = await runTask(session, [userText(opts.message)], {
mode,
signal: controller.signal,
renderer,
t,
});
// Task ended with an abort (LLM failure/reconnect exhausted/user interrupt): non-zero
// exit code, for scripts/CI to check.
if (result.aborted) process.exitCode = 1;
if (goalBudget !== null) {
// Goal mode: -m is the objective; the one run loops to a terminal state. Exit
// code follows the outcome — only a completed goal exits 0 (blocked /
// budget_limited / aborted are all "the goal did not finish", for scripts/CI to
// check).
const result = await runTask(session, [userText(opts.message)], {
mode,
signal: controller.signal,
renderer,
t,
goal: { budget: goalBudget, out },
});
if (result.goal?.outcome !== "complete") process.exitCode = 1;
} else {
const result = await runTask(session, [userText(opts.message)], {
mode,
signal: controller.signal,
renderer,
t,
});
// Task ended with an abort (LLM failure/reconnect exhausted/user interrupt): non-zero
// exit code, for scripts/CI to check.
if (result.aborted) process.exitCode = 1;
}
} finally {
process.off("SIGINT", onSigint);
session.dispose(); // Tear down managed long-running command sessions to avoid leaking background processes
+37
View File
@@ -0,0 +1,37 @@
/**
* Goal-command parsing (pure logic, shared by chat's `/goal` and run's `--goal`, unit-tested).
*
* Chat syntax: `/goal[:<budget>] <objective>` — the optional budget rides on the command
* token (`/goal:500k Raise coverage to 80%`); omitting it means no budget. Run passes the
* budget value (or `true` for a bare `--goal`), so only `parseTokenBudget` applies there.
*/
import { UNLIMITED_BUDGET } from "@prismshadow/penguin-core";
/**
* Parses a budget token: a positive number with an optional `k` / `m` suffix
* (`500k` = 500_000, `1.5m` = 1_500_000, `123456` literal). Returns null when invalid.
*/
export function parseTokenBudget(text: string): number | null {
const m = /^(\d+(?:\.\d+)?)([km])?$/i.exec(text.trim());
if (!m) return null;
const scale = m[2]?.toLowerCase() === "m" ? 1_000_000 : m[2]?.toLowerCase() === "k" ? 1_000 : 1;
const value = Math.round(Number(m[1]) * scale);
return value > 0 ? value : null;
}
export type GoalCommandResult =
| { ok: true; budget: number; objective: string }
| { ok: false; reason: "usage" }
| { ok: false; reason: "budget"; value: string };
/** Parses a full `/goal…` chat line (the caller has already matched the `/goal` prefix). */
export function parseGoalCommand(line: string): GoalCommandResult {
const m = /^\/goal(?::(\S+))?(?:\s+([\s\S]+))?$/.exec(line.trim());
if (!m) return { ok: false, reason: "usage" };
const rest = m[2]?.trim() ?? "";
if (!rest) return { ok: false, reason: "usage" };
if (m[1] === undefined) return { ok: true, budget: UNLIMITED_BUDGET, objective: rest };
const budget = parseTokenBudget(m[1]);
if (budget === null) return { ok: false, reason: "budget", value: m[1] };
return { ok: true, budget, objective: rest };
}
+60 -5
View File
@@ -63,7 +63,12 @@ export interface Messages {
vaultKey: string;
vaultValue: string;
};
run: { desc: string; message: string };
run: {
desc: string;
message: string;
/** run's --goal: goal mode, with an optional token budget value (`--goal 500k`). */
goal: string;
};
chat: { desc: string; resume: string };
serve: {
serverDesc: string;
@@ -161,6 +166,20 @@ export interface Messages {
compactionStop(mode: string, status: string, tokens?: { total: string; delta: string }): string;
/** Prompt shown when `/compact` has nothing to compact (session just started / two consecutive compactions). */
compactNothing(): string;
/** Dim line announcing one goal round (printed before the round runs). */
goalRound(round: number): string;
/** Dim summary line after a goal ends: how it ended, rounds run, tokens consumed. */
goalFinished(
outcome: "complete" | "blocked" | "budget_limited" | "aborted",
rounds: number,
tokens: string,
): string;
/** `/goal` usage error (missing objective / malformed command). */
goalUsage(): string;
/** Invalid token-budget value (chat `/goal:<budget>` or run `--goal <budget>`). */
goalBudgetInvalid(value: string): string;
/** run's --goal given an empty/whitespace -m (the objective must be non-empty text). */
goalObjectiveEmpty(): string;
/** Prompt for an invalid --approve mode. */
approveModeInvalid(value: string): string;
/** Render label for an approval decision (frontend renders the approval_decision event; one label each for allow/deny). */
@@ -278,7 +297,11 @@ const en: Messages = {
vaultKey: "Variable name (letters, digits and underscores; must not start with a digit)",
vaultValue: "Variable value, written to the Agent's agent_state/.vault.toml",
},
run: { desc: "Run a single Task", message: "Prompt for this Task" },
run: {
desc: "Run a single Task",
message: "Prompt for this Task",
goal: "Goal mode: loop until the goal completes; optional token budget (e.g. 500k, 2m)",
},
chat: {
desc: "Open the interactive REPL",
resume:
@@ -349,7 +372,7 @@ const en: Messages = {
header: headerEn,
chatHints: () =>
"Type a message to start a conversation; end a line with \\; typing while a task runs steers the agent; /compact to compact the context; /exit to quit; and Ctrl-C interrupts the current conversation.",
"Type a message to start a conversation; end a line with \\; typing while a task runs steers the agent; /goal runs a goal to completion; /compact to compact the context; /exit to quit; and Ctrl-C interrupts the current conversation.",
confirmExit: () => "Exit penguin? [y/N] ",
taskInterrupted: () => "[current conversation interrupted]",
steerQueued: (text) => `» steering queued (delivered with the next turn): ${text}`,
@@ -373,6 +396,20 @@ const en: Messages = {
: `[compaction] ${status}; keeping the current context`) +
(tokens ? ` · tokens ${tokens.total} (${tokens.delta})` : ""),
compactNothing: () => "[compaction] nothing to compact yet",
goalRound: (round) => `[goal] round ${round}`,
goalFinished: (outcome, rounds, tokens) => {
const label = {
complete: "completed",
blocked: "blocked (see the final reply for what it needs)",
budget_limited: "stopped: token budget exhausted",
aborted: "interrupted",
}[outcome];
return `[goal] ${label} · ${rounds} round${rounds === 1 ? "" : "s"} · tokens ${tokens}`;
},
goalUsage: () => "Usage: /goal[:<budget>] <objective> (e.g. /goal:500k fix all failing tests)",
goalBudgetInvalid: (value) =>
`Invalid token budget "${value}". Use a positive number with an optional k/m suffix (500k, 2m).`,
goalObjectiveEmpty: () => "Goal mode requires a non-empty objective: pass it via -m.",
approveModeInvalid: (value) =>
`Invalid approval mode "${value}". Use allow-all, deny-all, read-only, or always-ask.`,
approvalDecision: (decision) => (decision === "allow" ? "✓ [approved]" : "× [denied]"),
@@ -452,7 +489,11 @@ const zh: Messages = {
vaultKey: "变量名(字母、数字与下划线,不能以数字开头)",
vaultValue: "变量值,写入该 Agent 的 agent_state/.vault.toml",
},
run: { desc: "单次运行一个 Task", message: "本次 Task 的 Prompt" },
run: {
desc: "单次运行一个 Task",
message: "本次 Task 的 Prompt",
goal: "目标模式:循环运行直至目标完成;可选 token 预算(如 500k、2m)",
},
chat: {
desc: "打开交互式 REPL",
resume:
@@ -519,7 +560,7 @@ const zh: Messages = {
header: headerZh,
chatHints: () =>
"输入消息发起对话;行尾 \\ 续行;运行中输入可插话引导;/compact 压缩上下文;/exit 退出;Ctrl-C 中断对话。",
"输入消息发起对话;行尾 \\ 续行;运行中输入可插话引导;/goal 以目标模式运行至完成;/compact 压缩上下文;/exit 退出;Ctrl-C 中断对话。",
confirmExit: () => "确认退出 penguin?[y/N] ",
taskInterrupted: () => "[已中断当前对话]",
steerQueued: (text) => `» 插话已排队(随下一轮送达):${text}`,
@@ -543,6 +584,20 @@ const zh: Messages = {
: `[压缩] ${status === "aborted" ? "已中断" : "失败"},保留当前上下文`) +
(tokens ? ` · tokens ${tokens.total} (${tokens.delta})` : ""),
compactNothing: () => "[压缩] 当前上下文为空,无需压缩",
goalRound: (round) => `[目标] 第 ${round} 轮`,
goalFinished: (outcome, rounds, tokens) => {
const label = {
complete: "已完成",
blocked: "受阻(所缺条件见最后一条回复)",
budget_limited: "已停止:token 预算耗尽",
aborted: "已中断",
}[outcome];
return `[目标] ${label} · 共 ${rounds} 轮 · tokens ${tokens}`;
},
goalUsage: () => "用法:/goal[:<预算>] <目标>(例如 /goal:500k 修复所有失败的测试)",
goalBudgetInvalid: (value) =>
`无效的 token 预算 "${value}":应为正数,可带 k/m 后缀(500k、2m)。`,
goalObjectiveEmpty: () => "目标模式需要非空的目标文本:请通过 -m 传入。",
approveModeInvalid: (value) =>
`无效的审批模式 "${value}"。请使用 allow-all、deny-all、read-only 或 always-ask。`,
approvalDecision: (decision) => (decision === "allow" ? "✓ [已批准]" : "× [已拒绝]"),
+43 -5
View File
@@ -6,9 +6,15 @@
* it on allow, with execution possibly overlapping. The CLI only needs to consume the output
* stream and supply `approve`. The approval strategy is determined by the permission mode
* (allow-all / deny-all / read-only / always-ask per-call approval).
*
* Goal mode rides the same call (`opts.goal` → `session.run(prompt, { goal })`): core loops
* the rounds inside the one run, so this loop only adds the per-round rendering rhythm —
* a dim round line at each `[goal]` round boundary, per-round stats via `endTask`, and the
* outcome summary read from the stream's terminal `goal_finished` event.
*/
import { isEventMessage } from "@prismshadow/penguin-core";
import type { ApproveFn, OmniMessage, Session } from "@prismshadow/penguin-core";
import { goalFinishedOf, isEventMessage, isGoalRoundInput } from "@prismshadow/penguin-core";
import type { ApproveFn, GoalOutcome, OmniMessage, Session } from "@prismshadow/penguin-core";
import { dim, humanizeTokens } from "./render.js";
import type { StreamRenderer } from "./render.js";
import { makeApprove, promptApproval, type ApprovalMode } from "./approval.js";
import type { Messages } from "./i18n.js";
@@ -23,11 +29,18 @@ export interface RunTaskOptions {
interactivePrompt?: ApproveFn;
/** Message set. */
t: Messages;
/**
* Present = goal mode: the prompt's text is the objective and the one `session.run` loops
* until the goal reaches a terminal state (a single AbortSignal spans every round). `out`
* receives the dim round/summary lines the renderer doesn't own.
*/
goal?: { budget: number; out: NodeJS.WritableStream };
}
/** Result of one Task: `aborted` = the Task ended with an abort event (LLM failure/reconnect exhausted/user interrupt). */
/** Result of one Task: `aborted` = the Task ended with an abort event (LLM failure/reconnect exhausted/user interrupt); `goal` = the outcome of a goal-mode run (absent when the stream was cut off before the terminal event). */
export interface RunTaskResult {
aborted: boolean;
goal?: GoalOutcome;
}
export async function runTask(
@@ -88,20 +101,45 @@ export async function runTask(
// (auth errors, reconnect exhausted, etc.) into a main-session abort event rather than
// throwing; the result reported here reflects that, for `penguin run` to map to
// an exit code.
//
// In goal mode the round boundaries are the injected `[goal]` user messages core yields
// before each round: stats settle per round (the per-Task rhythm of a normal chat), so
// `segmentStartedAt` tracks the current round rather than the whole run.
const goal = opts.goal;
const startedAt = Date.now();
let segmentStartedAt = startedAt;
let aborted = false;
let round = 0;
let outcome: GoalOutcome | undefined;
try {
for await (const msg of session.run(prompt, {
approve,
...(opts.signal ? { signal: opts.signal } : {}),
...(goal ? { goal: { budget: goal.budget } } : {}),
})) {
if (isEventMessage(msg) && msg.payload.type === "abort" && (msg.origin?.length ?? 0) === 0) {
aborted = true;
}
if (goal) {
if (isGoalRoundInput(msg)) {
// Settle the previous round's stats before announcing the next (endTask is what
// prints the per-task `[stats]` line in a normal chat).
if (round > 0) opts.renderer.endTask(Date.now() - segmentStartedAt);
round++;
segmentStartedAt = Date.now();
goal.out.write(`${dim(opts.t.goalRound(round))}\n`);
}
outcome = goalFinishedOf(msg) ?? outcome;
}
opts.renderer.handle(msg);
}
} finally {
opts.renderer.endTask(Date.now() - startedAt);
opts.renderer.endTask(Date.now() - segmentStartedAt);
}
return { aborted };
if (goal && outcome) {
goal.out.write(
`${dim(opts.t.goalFinished(outcome.outcome, outcome.rounds, humanizeTokens(outcome.tokensUsed)))}\n`,
);
}
return { aborted, ...(outcome !== undefined ? { goal: outcome } : {}) };
}