feat(core,cli,server,web): add goal mode — loop Tasks on one Session until an objective completes (#66)

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
rank-Yu
2026-07-27 23:43:24 +08:00
committed by GitHub
parent e1141ca010
commit 46463bee26
61 changed files with 3411 additions and 188 deletions
+23 -2
View File
@@ -4,8 +4,9 @@
* penguin chat [--model-id <id> --provider <group>] [--project-id <id>] [--agent-id <id>]
* [--workspace <path>] [--approve <allow-all|deny-all|read-only|always-ask>]
*
* Each line of input starts one conversation turn; `/compact` proactively compacts the
* context (reason=manual); `/exit` or `/quit` exits.
* Each line of input starts one conversation turn; `/goal[:<budget>] <objective>` runs
* goal mode (looping until the goal reaches a terminal state);
* `/compact` proactively compacts the context (reason=manual); `/exit` or `/quit` exits.
* Uses the current directory when no Workspace is specified. A model reference is always an
* explicit `(provider, model_id)` pair, so `--model-id` and `--provider` must be given
* together; giving neither uses the Project's default model.
@@ -30,6 +31,7 @@ import { createAgent, userText, VERSION } from "@prismshadow/penguin-core";
import type { ApprovalDecision, OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core";
import { StreamRenderer, dim, renderHistory, sessionMetaTools } from "../render.js";
import { runTask } from "../task-loop.js";
import { parseGoalCommand } from "../goal-command.js";
import { parseApprovalAnswer, resolveApprovalMode } from "../approval.js";
import { LineComposer, PasteFilter } from "../input.js";
import type { Messages } from "../i18n.js";
@@ -375,6 +377,25 @@ export function registerChatCommand(program: Command, t: Messages): void {
renderer.endCompact(Date.now() - startedAt);
}
if (!sawMessage) out.write(`${t.compactNothing()}\n`);
} else if (text === "/goal" || text.startsWith("/goal:") || text.startsWith("/goal ")) {
// Goal mode: one command drives the whole loop; Ctrl-C aborts the entire
// goal (a single signal spans every round), never just the current round.
const parsed = parseGoalCommand(text);
if (!parsed.ok) {
const message =
parsed.reason === "budget" ? t.goalBudgetInvalid(parsed.value) : t.goalUsage();
out.write(`${t.error(message)}\n`);
} else {
resumable = true;
await runTask(session, [userText(parsed.objective)], {
mode,
signal: taskAbort.signal,
renderer,
interactivePrompt,
t,
goal: { budget: parsed.budget, out },
});
}
} else {
resumable = true;
await runTask(session, [userText(text)], {
+49 -10
View File
@@ -4,18 +4,23 @@
* penguin run -m <msg> [--model-id <id> --provider <group>] [--workspace <path>]
* [--project-id <id>] [--agent-id <id>]
* [--approve <allow-all|deny-all|read-only|always-ask>]
* [--goal [budget]]
*
* Uses the current directory when Workspace is unspecified; uses the Project's default model
* when model is unspecified. A model reference is always an explicit `(provider, model_id)`
* pair, so `--model-id` and `--provider` must be given together — giving only one of them is
* an error, never a lookup. Defaults to interactive per-call approval; `--approve`
* selects the permission mode.
* `--goal` switches to goal mode: `-m` becomes the objective and the run loops until the
* goal reaches a terminal state (optional value = token budget, e.g. `--goal 500k`); only a
* completed goal exits 0.
* Docs: /docs/cli § "penguin run".
*/
import type { Command } from "commander";
import { createAgent, userText, VERSION } from "@prismshadow/penguin-core";
import { UNLIMITED_BUDGET, createAgent, userText, VERSION } from "@prismshadow/penguin-core";
import { StreamRenderer, sessionMetaTools } from "../render.js";
import { runTask } from "../task-loop.js";
import { parseTokenBudget } from "../goal-command.js";
import { denyActivePrompt, resolveApprovalMode } from "../approval.js";
import type { Messages } from "../i18n.js";
@@ -30,6 +35,7 @@ export function registerRunCommand(program: Command, t: Messages): void {
.option("--agent-id <id>", t.common.agentId)
.option("--workspace <path>", t.common.workspace)
.option("--approve <mode>", t.common.approve)
.option("--goal [budget]", t.run.goal)
.action(async (opts) => {
// The model reference is a pair: commander can only require each option on its own,
// so the "both or neither" rule is enforced here. Giving neither is the normal case
@@ -39,6 +45,24 @@ export function registerRunCommand(program: Command, t: Messages): void {
process.exitCode = 1;
return;
}
// --goal's optional value is the token budget (`--goal 500k`); a bare --goal means no
// budget. Validated before any Session is created, like the model-pair check above.
let goalBudget: number | null = null;
if (opts.goal !== undefined) {
goalBudget = opts.goal === true ? UNLIMITED_BUDGET : parseTokenBudget(String(opts.goal));
if (goalBudget === null) {
process.stderr.write(`${t.error(t.goalBudgetInvalid(String(opts.goal)))}\n`);
process.exitCode = 1;
return;
}
// The objective must be non-empty text (core throws on an empty one — turn the
// programming-level error into a friendly refusal before any Session exists).
if (String(opts.message).trim() === "") {
process.stderr.write(`${t.error(t.goalObjectiveEmpty())}\n`);
process.exitCode = 1;
return;
}
}
const mode = resolveApprovalMode(opts.approve, t);
const agent = await createAgent({
@@ -70,15 +94,30 @@ export function registerRunCommand(program: Command, t: Messages): void {
// The assembled tool schemas decide each tool's call-line preview path (see render.ts).
renderer.useToolSchemas(sessionMetaTools(session));
try {
const result = await runTask(session, [userText(opts.message)], {
mode,
signal: controller.signal,
renderer,
t,
});
// Task ended with an abort (LLM failure/reconnect exhausted/user interrupt): non-zero
// exit code, for scripts/CI to check.
if (result.aborted) process.exitCode = 1;
if (goalBudget !== null) {
// Goal mode: -m is the objective; the one run loops to a terminal state. Exit
// code follows the outcome — only a completed goal exits 0 (blocked /
// budget_limited / aborted are all "the goal did not finish", for scripts/CI to
// check).
const result = await runTask(session, [userText(opts.message)], {
mode,
signal: controller.signal,
renderer,
t,
goal: { budget: goalBudget, out },
});
if (result.goal?.outcome !== "complete") process.exitCode = 1;
} else {
const result = await runTask(session, [userText(opts.message)], {
mode,
signal: controller.signal,
renderer,
t,
});
// Task ended with an abort (LLM failure/reconnect exhausted/user interrupt): non-zero
// exit code, for scripts/CI to check.
if (result.aborted) process.exitCode = 1;
}
} finally {
process.off("SIGINT", onSigint);
session.dispose(); // Tear down managed long-running command sessions to avoid leaking background processes