feat(core,cli,server,web): add goal mode — loop Tasks on one Session until an objective completes (#66)
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -4,8 +4,9 @@
|
||||
* penguin chat [--model-id <id> --provider <group>] [--project-id <id>] [--agent-id <id>]
|
||||
* [--workspace <path>] [--approve <allow-all|deny-all|read-only|always-ask>]
|
||||
*
|
||||
* Each line of input starts one conversation turn; `/compact` proactively compacts the
|
||||
* context (reason=manual); `/exit` or `/quit` exits.
|
||||
* Each line of input starts one conversation turn; `/goal[:<budget>] <objective>` runs
|
||||
* goal mode (looping until the goal reaches a terminal state);
|
||||
* `/compact` proactively compacts the context (reason=manual); `/exit` or `/quit` exits.
|
||||
* Uses the current directory when no Workspace is specified. A model reference is always an
|
||||
* explicit `(provider, model_id)` pair, so `--model-id` and `--provider` must be given
|
||||
* together; giving neither uses the Project's default model.
|
||||
@@ -30,6 +31,7 @@ import { createAgent, userText, VERSION } from "@prismshadow/penguin-core";
|
||||
import type { ApprovalDecision, OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core";
|
||||
import { StreamRenderer, dim, renderHistory, sessionMetaTools } from "../render.js";
|
||||
import { runTask } from "../task-loop.js";
|
||||
import { parseGoalCommand } from "../goal-command.js";
|
||||
import { parseApprovalAnswer, resolveApprovalMode } from "../approval.js";
|
||||
import { LineComposer, PasteFilter } from "../input.js";
|
||||
import type { Messages } from "../i18n.js";
|
||||
@@ -375,6 +377,25 @@ export function registerChatCommand(program: Command, t: Messages): void {
|
||||
renderer.endCompact(Date.now() - startedAt);
|
||||
}
|
||||
if (!sawMessage) out.write(`${t.compactNothing()}\n`);
|
||||
} else if (text === "/goal" || text.startsWith("/goal:") || text.startsWith("/goal ")) {
|
||||
// Goal mode: one command drives the whole loop; Ctrl-C aborts the entire
|
||||
// goal (a single signal spans every round), never just the current round.
|
||||
const parsed = parseGoalCommand(text);
|
||||
if (!parsed.ok) {
|
||||
const message =
|
||||
parsed.reason === "budget" ? t.goalBudgetInvalid(parsed.value) : t.goalUsage();
|
||||
out.write(`${t.error(message)}\n`);
|
||||
} else {
|
||||
resumable = true;
|
||||
await runTask(session, [userText(parsed.objective)], {
|
||||
mode,
|
||||
signal: taskAbort.signal,
|
||||
renderer,
|
||||
interactivePrompt,
|
||||
t,
|
||||
goal: { budget: parsed.budget, out },
|
||||
});
|
||||
}
|
||||
} else {
|
||||
resumable = true;
|
||||
await runTask(session, [userText(text)], {
|
||||
|
||||
@@ -4,18 +4,23 @@
|
||||
* penguin run -m <msg> [--model-id <id> --provider <group>] [--workspace <path>]
|
||||
* [--project-id <id>] [--agent-id <id>]
|
||||
* [--approve <allow-all|deny-all|read-only|always-ask>]
|
||||
* [--goal [budget]]
|
||||
*
|
||||
* Uses the current directory when Workspace is unspecified; uses the Project's default model
|
||||
* when model is unspecified. A model reference is always an explicit `(provider, model_id)`
|
||||
* pair, so `--model-id` and `--provider` must be given together — giving only one of them is
|
||||
* an error, never a lookup. Defaults to interactive per-call approval; `--approve`
|
||||
* selects the permission mode.
|
||||
* `--goal` switches to goal mode: `-m` becomes the objective and the run loops until the
|
||||
* goal reaches a terminal state (optional value = token budget, e.g. `--goal 500k`); only a
|
||||
* completed goal exits 0.
|
||||
* Docs: /docs/cli § "penguin run".
|
||||
*/
|
||||
import type { Command } from "commander";
|
||||
import { createAgent, userText, VERSION } from "@prismshadow/penguin-core";
|
||||
import { UNLIMITED_BUDGET, createAgent, userText, VERSION } from "@prismshadow/penguin-core";
|
||||
import { StreamRenderer, sessionMetaTools } from "../render.js";
|
||||
import { runTask } from "../task-loop.js";
|
||||
import { parseTokenBudget } from "../goal-command.js";
|
||||
import { denyActivePrompt, resolveApprovalMode } from "../approval.js";
|
||||
import type { Messages } from "../i18n.js";
|
||||
|
||||
@@ -30,6 +35,7 @@ export function registerRunCommand(program: Command, t: Messages): void {
|
||||
.option("--agent-id <id>", t.common.agentId)
|
||||
.option("--workspace <path>", t.common.workspace)
|
||||
.option("--approve <mode>", t.common.approve)
|
||||
.option("--goal [budget]", t.run.goal)
|
||||
.action(async (opts) => {
|
||||
// The model reference is a pair: commander can only require each option on its own,
|
||||
// so the "both or neither" rule is enforced here. Giving neither is the normal case
|
||||
@@ -39,6 +45,24 @@ export function registerRunCommand(program: Command, t: Messages): void {
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
// --goal's optional value is the token budget (`--goal 500k`); a bare --goal means no
|
||||
// budget. Validated before any Session is created, like the model-pair check above.
|
||||
let goalBudget: number | null = null;
|
||||
if (opts.goal !== undefined) {
|
||||
goalBudget = opts.goal === true ? UNLIMITED_BUDGET : parseTokenBudget(String(opts.goal));
|
||||
if (goalBudget === null) {
|
||||
process.stderr.write(`${t.error(t.goalBudgetInvalid(String(opts.goal)))}\n`);
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
// The objective must be non-empty text (core throws on an empty one — turn the
|
||||
// programming-level error into a friendly refusal before any Session exists).
|
||||
if (String(opts.message).trim() === "") {
|
||||
process.stderr.write(`${t.error(t.goalObjectiveEmpty())}\n`);
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
}
|
||||
const mode = resolveApprovalMode(opts.approve, t);
|
||||
|
||||
const agent = await createAgent({
|
||||
@@ -70,15 +94,30 @@ export function registerRunCommand(program: Command, t: Messages): void {
|
||||
// The assembled tool schemas decide each tool's call-line preview path (see render.ts).
|
||||
renderer.useToolSchemas(sessionMetaTools(session));
|
||||
try {
|
||||
const result = await runTask(session, [userText(opts.message)], {
|
||||
mode,
|
||||
signal: controller.signal,
|
||||
renderer,
|
||||
t,
|
||||
});
|
||||
// Task ended with an abort (LLM failure/reconnect exhausted/user interrupt): non-zero
|
||||
// exit code, for scripts/CI to check.
|
||||
if (result.aborted) process.exitCode = 1;
|
||||
if (goalBudget !== null) {
|
||||
// Goal mode: -m is the objective; the one run loops to a terminal state. Exit
|
||||
// code follows the outcome — only a completed goal exits 0 (blocked /
|
||||
// budget_limited / aborted are all "the goal did not finish", for scripts/CI to
|
||||
// check).
|
||||
const result = await runTask(session, [userText(opts.message)], {
|
||||
mode,
|
||||
signal: controller.signal,
|
||||
renderer,
|
||||
t,
|
||||
goal: { budget: goalBudget, out },
|
||||
});
|
||||
if (result.goal?.outcome !== "complete") process.exitCode = 1;
|
||||
} else {
|
||||
const result = await runTask(session, [userText(opts.message)], {
|
||||
mode,
|
||||
signal: controller.signal,
|
||||
renderer,
|
||||
t,
|
||||
});
|
||||
// Task ended with an abort (LLM failure/reconnect exhausted/user interrupt): non-zero
|
||||
// exit code, for scripts/CI to check.
|
||||
if (result.aborted) process.exitCode = 1;
|
||||
}
|
||||
} finally {
|
||||
process.off("SIGINT", onSigint);
|
||||
session.dispose(); // Tear down managed long-running command sessions to avoid leaking background processes
|
||||
|
||||
Reference in New Issue
Block a user