perf(server,web): cursor-paginate session history with tail-first loading (#202)

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Yaowei Zheng
2026-08-05 00:16:25 +08:00
committed by GitHub
parent fc7cb5b8ed
commit dd8ddbb69b
22 changed files with 2153 additions and 80 deletions
+18 -3
View File
@@ -306,17 +306,32 @@ export const patchSession = (sessionId: string, body: SessionPatchRequest) =>
export const deleteSession = (sessionId: string) =>
apiFetch<void>(`/api/sessions/${encodeURIComponent(sessionId)}`, { method: "DELETE" });
/** Windowed history request: the newest N units (tail), or the N units before a cursor. */
export type MessagesPageQuery =
{ kind: "tail"; limit: number } | { kind: "before"; cursor: string; limit: number };
/**
* History rebuild. Carries the server's clock at read time (see ApiFetchMeta.serverNowMs)
* alongside the messages: a Task still running has no Trace entry for the event currently in
* flight, so its elapsed can only be measured by differencing this against the Task's first
* message timestamp — both server-side values, so no client clock offset enters the result
* (see pushMessages).
*
* With `page`, requests a WINDOW instead of the full transcript (tail-first loading /
* scroll-up backfill — see stream-controller): the response then carries
* `MessagesResponse.page`. Omitted = the legacy full read (the resync fallback path).
*/
export const getMessages = (sessionId: string) =>
apiFetchWithMeta<MessagesResponse>(
`/api/sessions/${encodeURIComponent(sessionId)}/messages`,
export const getMessages = (sessionId: string, page?: MessagesPageQuery) => {
const qs =
page === undefined
? ""
: page.kind === "tail"
? `?tailLimit=${page.limit}`
: `?before=${encodeURIComponent(page.cursor)}&limit=${page.limit}`;
return apiFetchWithMeta<MessagesResponse>(
`/api/sessions/${encodeURIComponent(sessionId)}/messages${qs}`,
).then(({ data, serverNowMs }) => ({ ...data, serverNowMs }));
};
// Task execution, approval, abort, compaction ------------------------------------------------------
+50 -5
View File
@@ -289,16 +289,43 @@ export function ChatPage() {
discard: discardSessionDraft,
} = useSessionDraft(selected?.sessionId ?? null);
// The rendered transcript: backfilled older windows (frozen items, negative ids) ahead
// of the live tail model's items. Version keys the memo — a prepend bumps it.
const allItems = useMemo(
() =>
stream.prefixItems.length > 0
? [...stream.prefixItems, ...stream.model.items]
: stream.model.items,
// eslint-disable-next-line react-hooks/exhaustive-deps
[stream.version, routeSessionId],
);
// Derivations over the stream items (the model mutates in place, so `version` — its own
// repaint signal — keys the memos; the session id covers a switch racing a same-valued
// version): the composer's ↑/↓ recall list and the left outline's entries.
// version): the composer's ↑/↓ recall list and the left outline's entries. Both read the
// LOADED transcript (prefix included), so backfilling extends recall and the index alike.
const inputHistory = useMemo(
() => buildInputHistory(stream.model.items),
() => buildInputHistory(allItems),
// eslint-disable-next-line react-hooks/exhaustive-deps
[stream.version, routeSessionId],
);
const outline = useMemo(
() => buildOutline(stream.model.items),
() => buildOutline(allItems),
// eslint-disable-next-line react-hooks/exhaustive-deps
[stream.version, routeSessionId],
);
// The subagents panel's model view: the live model, with backfilled windows' items and
// nested subagent models merged in — a chip clicked on a backfilled turn must still
// resolve its historical Task slice and child conversation. Scalar fields snapshot per
// version, which is exactly as fresh as everything else the panel renders.
const panelModel = useMemo<StreamModel>(
() =>
stream.prefixItems.length > 0 || stream.prefixSubagents.size > 0
? {
...stream.model,
items: [...allItems],
subagents: new Map([...stream.prefixSubagents, ...stream.model.subagents]),
}
: stream.model,
// eslint-disable-next-line react-hooks/exhaustive-deps
[stream.version, routeSessionId],
);
@@ -347,6 +374,10 @@ export function ChatPage() {
// dot still signal), and never re-triggers an already-open panel (that would yank a pinned
// historical graph back to the latest Task); the tracker consumes the attempt regardless.
const panelTaskScopeRef = useRef(createPanelTaskScope());
// Deliberately the LIVE model's items only (never the backfilled prefix): the tracker
// reads an INCREASE as "the user started a new Task", and a scroll-up backfill growing
// the count would spuriously re-arm the auto-open mid-conversation. The latest Task
// always lives in the live tail window, so live-only loses nothing.
const taskCount = taskStartCount(stream.model.items);
const liveSpawn = stream.taskState !== "idle" && latestTaskHasSubagent(stream.model);
useEffect(() => {
@@ -1087,6 +1118,7 @@ export function ChatPage() {
{!railFit.shown && (
<OutlineMenuButton
entries={outline}
turnOffset={stream.outlineOffset}
scrollRef={streamScrollRef}
running={stream.taskState !== "idle"}
/>
@@ -1240,16 +1272,27 @@ export function ChatPage() {
</div>
) : (
<MessageStream
items={stream.model.items}
items={allItems}
version={stream.version}
ctx={ctx}
scrollElRef={streamScrollRef}
// Scroll-up backfill of older history windows (tail-first
// loading): near-top scrolling prepends the previous window,
// scroll position anchored (see MessageStream).
older={{
hasMore: stream.older.hasMore,
loading: stream.older.loading,
error: stream.older.error,
prependedCount: stream.prefixItems.length,
onLoad: stream.loadOlder,
}}
// Tick-rail minimap over the stream's left gutter (zero layout
// width; hides itself when the gutter is too narrow or the
// pointer can't hover).
outline={
<ConversationOutline
entries={outline}
turnOffset={stream.outlineOffset}
version={stream.version}
scrollRef={streamScrollRef}
running={stream.taskState !== "idle"}
@@ -1289,7 +1332,9 @@ export function ChatPage() {
<SubagentsPanel
session={selected}
panel={subagentsPanel}
model={stream.model}
// The merged view (backfilled windows included): a chip on an older turn keeps
// its historical graph and child conversation reachable after pagination.
model={panelModel}
version={stream.version}
taskRunning={stream.taskState !== "idle"}
ctx={ctx}
@@ -41,7 +41,8 @@ import { Dropdown } from "../../components/ui/dropdown";
import { GlyphIcon } from "../../components/ui/glyph-icon";
import type { OutlineEntry } from "./outline-model";
import {
OUTLINE_MIN_TURNS,
globalTurnNumber,
outlineVisible,
previewText,
railTickPitch,
railWindowHalf,
@@ -202,12 +203,20 @@ function answerPreview(
export function ConversationOutline({
entries,
turnOffset = 0,
version,
scrollRef,
running,
fit,
}: {
entries: OutlineEntry[];
/**
* Turns that exist BEFORE the loaded history window (windowed loading): added to every
* tick's global turn number, counted toward the visibility gate, and >0 keeps the
* "more above" edge dots on — numbering never lies about unloaded turns, and the rail
* signals that scrolling up will reveal them.
*/
turnOffset?: number;
/** Stream repaint signal: re-runs the scrollspy as content grows or the stream remounts. */
version: number;
/** MessageStream's scroll container (null while the stream isn't mounted, e.g. the empty greeting). */
@@ -227,7 +236,7 @@ export function ConversationOutline({
// cheap, and keying on version also re-binds after the stream remounts on a session switch.
useEffect(() => {
const el = scrollRef.current;
if (!fit.shown || !el || entries.length < OUTLINE_MIN_TURNS) return;
if (!fit.shown || !el || !outlineVisible(turnOffset, entries.length)) return;
const ids = new Set(entries.map((entry) => entry.anchorId));
let raf: number | null = null;
const compute = () => {
@@ -246,9 +255,11 @@ export function ConversationOutline({
};
// `entries` is rebuilt per version; length + version cover it.
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [fit.shown, scrollRef, entries.length, version]);
}, [fit.shown, scrollRef, entries.length, turnOffset, version]);
if (!fit.shown || entries.length < OUTLINE_MIN_TURNS) return null;
// The gate counts the WHOLE conversation (loaded + unloaded turns): a long session
// whose tail window happens to hold few entries still deserves its index.
if (!fit.shown || !outlineVisible(turnOffset, entries.length)) return null;
/** Tick center Y relative to the rail overlay (the preview card anchors to it, clamped in render). */
const tickTop = (e: MouseEvent<HTMLElement> | FocusEvent<HTMLElement>) => {
@@ -291,7 +302,9 @@ export function ConversationOutline({
inside the rail by construction, and this guarantees a mismeasure still can't
push ticks (which take pointer events) out over the toolbar or composer. */}
<div className="max-h-full overflow-hidden">
{start > 0 && <RailOverflowMark edge="above" />}
{/* "More above" also covers turns not yet LOADED (turnOffset > 0): the dots tell
the same story either way — earlier turns exist beyond the rendered ticks. */}
{(start > 0 || turnOffset > 0) && <RailOverflowMark edge="above" />}
{visible.map((entry, i) => {
const active = entry.anchorId === activeId;
return (
@@ -300,10 +313,10 @@ export function ConversationOutline({
type="button"
data-outline-tick={entry.anchorId}
aria-current={active || undefined}
// Global turn number (start + i): the window changes which ticks render,
// never how a turn is numbered.
// Global turn number (turnOffset + start + i): neither the sliding window
// nor a partially-loaded history changes how a turn is numbered.
aria-label={S.chat.outlineTickLabel(
start + i + 1,
globalTurnNumber(turnOffset, start + i),
entry.question || S.chat.outlineNoText,
)}
style={{ height: pitch }}
@@ -373,17 +386,20 @@ export function ConversationOutline({
*/
export function OutlineMenuButton({
entries,
turnOffset = 0,
scrollRef,
running,
}: {
entries: OutlineEntry[];
/** Turns before the loaded window (windowed loading): gates visibility on the WHOLE conversation; the list itself shows loaded turns only. */
turnOffset?: number;
scrollRef: RefObject<HTMLDivElement | null>;
running: boolean;
}) {
const [open, setOpen] = useState(false);
const [activeId, setActiveId] = useState<number | null>(null);
// Same gate as the rail: with this few turns neither shape earns its place.
if (entries.length < OUTLINE_MIN_TURNS) return null;
if (!outlineVisible(turnOffset, entries.length)) return null;
const setOpenComputing = (next: boolean) => {
if (next && scrollRef.current) {
@@ -155,12 +155,29 @@ export function MessageItems({ items, ctx }: { items: ChatItem[]; ctx: StreamRen
return <>{nodes}</>;
}
/** Scroll-up backfill wiring (windowed history): state + trigger for the top-of-stream affordance. */
export interface OlderHistoryControls {
/** Older windows exist beyond the loaded history (scrolling near the top triggers onLoad). */
hasMore: boolean;
/** A backfill request is in flight (spinner row). */
loading: boolean;
/** The last backfill failed (retry row); null = fine. */
error: string | null;
/** Number of windows already prepended: the prepend signal for scroll anchoring, and >0 gates the beginning-of-history marker (a session that fit one window shows no extra chrome). */
prependedCount: number;
onLoad: () => void;
}
/** Distance from the top (px) at which scrolling starts fetching the previous history window. */
const OLDER_TRIGGER_PX = 300;
export function MessageStream({
items,
version,
ctx,
scrollElRef,
outline,
older,
}: {
items: ChatItem[];
/** View-model version number (a repaint signal for in-place updates that also drives auto-scroll). */
@@ -174,6 +191,8 @@ export function MessageStream({
* and anchor its absolute positioning to this wrapper, which only this component owns.
*/
outline?: ReactNode;
/** Scroll-up backfill of older history windows; omitted = the whole transcript is loaded (no top affordance). */
older?: OlderHistoryControls;
}) {
const scrollRef = useRef<HTMLDivElement>(null);
// An upward-swipe intent immediately exits auto-follow; scrolling back near the bottom resumes it — see stream-follow.ts (#75) for the exact rule.
@@ -220,6 +239,19 @@ export function MessageStream({
);
};
/** Held refs for prepend scroll anchoring (see the layout effect below). */
const olderRef = useRef(older);
olderRef.current = older;
const lastPrependedRef = useRef(older?.prependedCount ?? 0);
const lastHeightRef = useRef(0);
/** Near the top of loaded history: fetch the previous window (loading/error states gate re-triggering; the retry row is click-driven). */
const maybeLoadOlder = (el: HTMLDivElement) => {
const o = olderRef.current;
if (!o || !o.hasMore || o.loading || o.error !== null) return;
if (el.scrollTop < OLDER_TRIGGER_PX) o.onLoad();
};
const onScroll = () => {
const el = scrollRef.current;
if (!el) return;
@@ -234,6 +266,7 @@ export function MessageStream({
scrollHeight: el.scrollHeight,
clientHeight: el.clientHeight,
});
maybeLoadOlder(el);
syncJump();
};
@@ -242,11 +275,24 @@ export function MessageStream({
// during the animated return — the glide owns the scroll position until it arrives.
useLayoutEffect(() => {
const el = scrollRef.current;
// Prepend scroll anchoring: when older windows land ABOVE the viewport, keep the
// message the user was reading exactly where it was by offsetting scrollTop by the
// content growth (same pre-paint timing as the stick snap, so nothing flashes).
// Keyed on the prepend count — ordinary streaming growth at the bottom must not
// shift the view. lastHeightRef is refreshed every commit, so at the prepend commit
// it still holds the pre-prepend height. Skipped while sticking (the snap below
// owns the position; a prepend while stuck at the bottom cannot move the tail).
const prepended = older?.prependedCount ?? 0;
if (el && prepended > lastPrependedRef.current && !follow.stick && !returningRef.current) {
el.scrollTop += el.scrollHeight - lastHeightRef.current;
}
lastPrependedRef.current = prepended;
if (el) lastHeightRef.current = el.scrollHeight;
if (el && follow.stick && !returningRef.current) el.scrollTop = el.scrollHeight;
syncJump();
// syncJump is recreated per render; the effect intentionally keys on stream growth only.
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [version, follow]);
}, [version, follow, older?.prependedCount]);
/** Back-to-bottom: glide down to the live bottom (reduced motion gets an instant jump); follow re-engages on arrival. */
const jumpToLatest = () => {
@@ -316,6 +362,32 @@ export function MessageStream({
className="anim-fade h-full overflow-y-auto px-4 py-4 md:px-6"
>
<div className="mx-auto max-w-3xl">
{/* Top-of-history affordance: spinner while the previous window loads, a click-to-retry
row after a failure, and — once at least one window was backfilled — a quiet
beginning-of-conversation marker when there is nothing older. Idle-with-more shows
nothing: scrolling near the top triggers the fetch by itself. */}
{older && items.length > 0 && (
<div className="flex justify-center pb-2">
{older.loading ? (
<span className="flex items-center gap-2 py-1 text-xs text-gray-400 dark:text-gray-500">
<span className="inline-block h-3 w-3 animate-spin rounded-full border-2 border-gray-400 border-t-transparent" />
{S.chat.loadingEarlier}
</span>
) : older.error !== null ? (
<button
type="button"
onClick={older.onLoad}
className="py-1 text-xs text-red-600 transition-colors duration-150 hover:text-red-700 dark:text-red-400 dark:hover:text-red-300"
>
{S.chat.loadEarlierRetry}
</button>
) : !older.hasMore && older.prependedCount > 0 ? (
<span className="py-1 text-xs text-gray-400 dark:text-gray-500">
{S.chat.historyBeginning}
</span>
) : null}
</div>
)}
{items.length === 0 ? (
<EmptyState title={S.chat.emptyStream} />
) : (
@@ -36,6 +36,24 @@ const ANSWER_CAP = 500;
*/
export const OUTLINE_MIN_TURNS = 5;
/**
* With windowed history loading, entries built from the loaded items are only the tail of
* the conversation: `turnOffset` (MessagesPageInfo.earlierTurns — the server counts with
* the same entry rule buildOutline applies; the two must stay in step) is the number of
* turns that exist before them. These two helpers are the ONLY places that combine the
* offset with loaded-entry indices, so numbering can never drift between the shapes.
*/
/** Global 1-based turn number of loaded entry `index` (its position in the loaded entries array). */
export function globalTurnNumber(turnOffset: number, index: number): number {
return turnOffset + index + 1;
}
/** Whether the outline earns its place: the gate counts the WHOLE conversation — unloaded earlier turns included — never just the loaded window. */
export function outlineVisible(turnOffset: number, loadedCount: number): boolean {
return turnOffset + loadedCount >= OUTLINE_MIN_TURNS;
}
/** Rail window half-widths: at most this many ticks render before/after the active one. */
export const OUTLINE_WINDOW_BEFORE = 20;
export const OUTLINE_WINDOW_AFTER = 20;
@@ -24,19 +24,37 @@ import type {
import { getGoal, getMe, getMessages } from "../../api/endpoints";
import { openSessionStream } from "../../api/sse";
import { createStreamController } from "../../lib/omni/stream-controller";
import type { PendingApproval, StreamController } from "../../lib/omni/stream-controller";
import type {
OlderHistoryState,
PendingApproval,
StreamController,
} from "../../lib/omni/stream-controller";
import { createStreamModel } from "../../lib/omni/stream-model";
import type { StreamModel } from "../../lib/omni/stream-model";
import type { ChatItem, StreamModel } from "../../lib/omni/stream-model";
import type { GoalBannerState } from "./goal-use";
export type { PendingApproval } from "../../lib/omni/stream-controller";
export type { OlderHistoryState, PendingApproval } from "../../lib/omni/stream-controller";
/** Minimum spacing between version commits: below one frame at 8fps, invisible as staleness, but ~8× fewer full re-parses of the growing message during a fast large-code stream. */
const BUMP_MIN_INTERVAL_MS = 120;
export interface SessionStreamState {
/** View model (updated in place; version bump triggers re-render). */
/** View model (updated in place; version bump triggers re-render): the LIVE tail window. */
model: StreamModel;
/**
* Backfilled older-window items, oldest first — the transcript renders
* `[...prefixItems, ...model.items]`. Empty until the user scrolls up past the tail
* window (see loadOlder); ids are negative and unique, so keys/anchors stay clean.
*/
prefixItems: readonly ChatItem[];
/** Nested subagent models owned by backfilled windows (the subagents panel merges them with model.subagents). */
prefixSubagents: ReadonlyMap<string, StreamModel>;
/** Outline entries existing before the oldest loaded window (global turn-numbering offset). */
outlineOffset: number;
/** Scroll-up backfill state (top affordance: spinner / retry / beginning-of-history). */
older: OlderHistoryState;
/** Prepend the previous history window (triggered near the top of the loaded transcript). */
loadOlder: () => void;
version: number;
/** True until history finishes loading. */
loading: boolean;
@@ -79,6 +97,9 @@ export interface SessionStreamState {
}
const EMPTY_PENDING: ReadonlyMap<string, PendingApproval> = new Map();
const EMPTY_PREFIX: readonly ChatItem[] = [];
const EMPTY_SUBAGENTS: ReadonlyMap<string, StreamModel> = new Map();
const IDLE_OLDER: OlderHistoryState = { hasMore: false, loading: false, error: null };
export function useSessionStream(
sessionId: string | null,
@@ -208,8 +229,9 @@ export function useSessionStream(
const controller = createStreamController({
// The whole response rides through: `live` (in-progress stream tail) lets the
// controller seed the currently streaming message after a reload (see stream-controller).
loadMessages: () => getMessages(sessionId),
// controller seed the currently streaming message after a reload, and `page` (the
// windowed-history envelope) drives tail-first loading (see stream-controller).
loadMessages: (page) => getMessages(sessionId, page),
onTaskState: setTaskState,
onQueuedFollowUps: setQueuedFollowUps,
onPendingSteering: setPendingSteering,
@@ -269,6 +291,10 @@ export function useSessionStream(
void controllerRef.current?.retry();
}, []);
const loadOlder = useCallback(() => {
void controllerRef.current?.loadOlder();
}, []);
const dismissModelAuthDead = useCallback(() => {
const m = controllerRef.current?.model;
if (m && m.lastAuthFailureMs !== null) {
@@ -282,6 +308,11 @@ export function useSessionStream(
return {
model: controllerRef.current?.model ?? placeholderRef.current,
prefixItems: controllerRef.current?.prefixItems ?? EMPTY_PREFIX,
prefixSubagents: controllerRef.current?.prefixSubagents ?? EMPTY_SUBAGENTS,
outlineOffset: controllerRef.current?.outlineOffset ?? 0,
older: controllerRef.current?.older ?? IDLE_OLDER,
loadOlder,
version,
loading,
taskState,
+232 -10
View File
@@ -34,6 +34,7 @@ import type { OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core/omn
import type {
GoalServerEvent,
MessagesLiveTail,
MessagesPageInfo,
PendingSteeringInfo,
ServerEvent,
SessionStatus,
@@ -51,7 +52,8 @@ import {
pushMessages,
registerLocalDecision,
} from "./stream-model";
import type { StreamModel } from "./stream-model";
import type { ChatItem, StreamModel } from "./stream-model";
import { seedPriorStats } from "./task-stats";
/** A single pending approval (keyed by approvalKey(origin, toolCallId)). */
export interface PendingApproval {
@@ -73,17 +75,45 @@ function parseEventId(id: string): { epoch: string; seq: number } | null {
return { epoch: id.slice(0, sep), seq };
}
/** A windowed history request (mirrors the server's tailLimit / before params). */
export type MessagesPageQuery =
{ kind: "tail"; limit: number } | { kind: "before"; cursor: string; limit: number };
/**
* Initial (tail) window size, in message-bearing units — one unit = one Task, opened by
* a user prompt (the server's cut rule; see MessagesPageInfo). 200 covers the vast
* majority of real sessions in a single request, so ordinary conversations still load
* whole exactly as before — only the pathological long tail (months-long sessions,
* agentic marathons) starts windowed, which is the point: their full-transcript reads
* were the unbounded memory/disk cost this pagination removes.
*/
export const TAIL_UNITS = 200;
/** Scroll-up backfill window size: smaller than the tail so each prepend stays snappy. */
export const OLDER_UNITS = 100;
/**
* Item-id space reserved per prepended window. The live model numbers its items upward
* from 1; each prepended window numbers upward from its own NEGATIVE base, so ids stay
* unique across the concatenated view (React keys, outline anchors) without ever
* renumbering already-mounted items. A window holds at most a few thousand items —
* far under the span.
*/
const PREPEND_ID_SPAN = 1_000_000;
export interface StreamControllerDeps {
/**
* Fetch history messages (GET /api/sessions/:id/messages), including the live tail while
* running. `serverNowMs` is the server's clock at read time (the response's `Date` header);
* omitted/null just costs a running Task's header the time its in-flight event has taken so
* far, which falls back to the Trace's own span (see pushMessages).
* far, which falls back to the Trace's own span (see pushMessages). `page` requests a
* WINDOW (the response then carries `page`); omitted = the legacy full transcript.
*/
loadMessages: () => Promise<{
loadMessages: (page?: MessagesPageQuery) => Promise<{
messages: OmniMessage[];
live?: MessagesLiveTail;
serverNowMs?: number | null;
page?: MessagesPageInfo;
}>;
/** Authoritative running state from the stream (covers both the subscription snapshot and transition events). */
onTaskState: (state: SessionStatus) => void;
@@ -108,14 +138,37 @@ export interface StreamControllerDeps {
now?: () => number;
}
/** Scroll-up backfill state (drives the stream's top affordance). */
export interface OlderHistoryState {
/** Older windows exist beyond the loaded prefix. */
hasMore: boolean;
/** A backfill request is in flight. */
loading: boolean;
/** The last backfill failed (the affordance offers a retry); null = fine. */
error: string | null;
}
export interface StreamController {
/** The current view model (a resync rebuild swaps in a new object). */
/** The current view model (a resync rebuild swaps in a new object): the LIVE tail window. */
readonly model: StreamModel;
/**
* Items of the backfilled older windows, oldest first — render them immediately BEFORE
* `model.items`. Frozen once built (their Tasks are complete); item ids are negative
* and unique across windows, so the concatenated list keys/anchors cleanly.
*/
readonly prefixItems: readonly ChatItem[];
/** Nested subagent models of the backfilled windows (merged view for the subagents panel; disjoint from model.subagents — a child session lives in exactly one window). */
readonly prefixSubagents: ReadonlyMap<string, StreamModel>;
/** Outline entries that exist before the OLDEST loaded window: the outline's global numbering offset. */
readonly outlineOffset: number;
readonly older: OlderHistoryState;
readonly pendingApprovals: ReadonlyMap<string, PendingApproval>;
/** Load history for the first time (called once after connect-first). */
/** Load history for the first time (called once after connect-first): fetches the TAIL window. */
load: () => Promise<void>;
/** Retry entry point after a history load failure (keeps the buffer, refetches history). */
retry: () => Promise<void>;
/** Prepend the previous window (scroll-up backfill); no-op while loading, failed, at the beginning, or before the initial load settled. */
loadOlder: () => Promise<void>;
/** SSE OmniMessage entry point (`eventId`: the SSE event id, used for live-tail cursor alignment). */
handleOmni: (msg: OmniMessage, eventId?: string | null) => void;
/** SSE server-event entry point (`eventId`: same as handleOmni). */
@@ -146,6 +199,49 @@ export function createStreamController(deps: StreamControllerDeps): StreamContro
/** Whether the most recent load failed (retry only takes effect after a failure, to avoid mistakenly replaying history). */
let failed = false;
// —— Windowed-history state (tail-first load + scroll-up backfill) ——
/** Items of backfilled older windows, oldest first (frozen; rendered before model.items). */
let prefixItems: ChatItem[] = [];
/** Nested subagent models owned by backfilled windows. */
let prefixSubagents = new Map<string, StreamModel>();
/** How many windows have been prepended (derives each one's negative item-id base). */
let prependCount = 0;
/** Cursor for the NEXT older window (= the oldest loaded window's start); null = beginning reached or full transcript loaded. */
let nextBefore: string | null = null;
/**
* The LIVE tail window's start cursor, as returned by the last tail fetch — the resync
* continuity anchor: a refetched tail whose start cursor equals this provably abuts the
* retained prefix. Null = the tail reaches the beginning (or the transcript was loaded
* whole), in which case there is no prefix to splice against.
*/
let tailStartCursor: string | null = null;
const older: OlderHistoryState = { hasMore: false, loading: false, error: null };
/** Outline entries before the OLDEST loaded window (the outline's numbering offset). */
let outlineOffset = 0;
/** Reset all windowed-history bookkeeping to "everything loaded from the beginning". */
const resetPaging = (): void => {
prefixItems = [];
prefixSubagents = new Map();
prependCount = 0;
nextBefore = null;
tailStartCursor = null;
older.hasMore = false;
older.loading = false;
older.error = null;
outlineOffset = 0;
};
/** Adopt a TAIL page's pagination envelope as the fresh baseline (initial load / prefix-dropping rebuild). */
const adoptTailPage = (page: MessagesPageInfo | undefined): void => {
resetPaging();
if (page === undefined) return; // full transcript: nothing older exists by definition
nextBefore = page.before ?? null;
tailStartCursor = page.before ?? null;
older.hasMore = page.before !== undefined;
outlineOffset = page.earlierTurns;
};
const clearPending = (): void => {
if (pending.size === 0) return;
pending.clear();
@@ -222,6 +318,32 @@ export function createStreamController(deps: StreamControllerDeps): StreamContro
}
};
/**
* resync_required decision tree — which refetch rebuilds the model. Resync means the
* SSE buffer was evicted and the transcript state is suspect, so every branch below
* chooses correctness over cleverness (ANY doubt falls back to the full read):
*
* 1. No backfilled prefix → refetch the TAIL window. Identical in shape to the
* initial load: the window is a disk-true suffix, the buffered events replay
* with overlap dedup, and the live attachment weaves in under the existing
* channel-epoch guard (weaveLiveTail skips seeding when the cursor's epoch
* doesn't match the events seen on this connection).
* 2. Prefix retained → refetch the TAIL window and splice ONLY when continuity is
* provable: the refetched window's start cursor must EQUAL the recorded start
* of the current tail window (cursors are (shard, ordinal) positions on
* immutable storage, so equality proves the new tail abuts the prefix exactly —
* no gap, no overlap). Equality holds precisely when no new unit started since
* the last tail fetch, the common mid-Task resync.
* 3. Prefix retained but the refetched tail reaches the very beginning (no cursor)
* → the tail alone provably covers everything: drop the prefix and use it.
* 4. Anything else — the start cursor moved (new units arrived), the response
* carried no page envelope, or the tail fetch itself failed mid-decision —
* is doubt: fall back to the legacy FULL refetch (one complete transcript, no
* prefix, offsets zeroed). Slow but beyond suspicion.
*
* The decision runs inside load() (it needs the response); this entry only picks the
* request shape.
*/
const rebuild = async (): Promise<void> => {
epoch += 1;
phase = "buffering";
@@ -238,7 +360,10 @@ export function createStreamController(deps: StreamControllerDeps): StreamContro
// the input for a skeleton, losing scroll position, expanded tool cards, and composer
// focus/draft.) Deltas arriving during the refetch are buffered and replayed on swap; the brief
// no-new-text pause is invisible next to a full teardown.
await load(epoch, createStreamModel(localDecisions));
await load(epoch, createStreamModel(localDecisions), {
page: { kind: "tail", limit: TAIL_UNITS },
splice: prefixItems.length > 0,
});
};
/**
@@ -274,15 +399,45 @@ export function createStreamController(deps: StreamControllerDeps): StreamContro
return [...pre, ...seeds, ...post];
};
const load = async (currentEpoch: number, freshModel?: StreamModel): Promise<void> => {
const load = async (
currentEpoch: number,
freshModel?: StreamModel,
opts: { page?: MessagesPageQuery; splice?: boolean } = {},
): Promise<void> => {
try {
const { messages, live, serverNowMs } = await deps.loadMessages();
let res = await deps.loadMessages(opts.page);
if (disposed || currentEpoch !== epoch) return;
if (opts.splice === true) {
// The resync decision tree's prefix-retained branches (see rebuild): splice only
// on exact cursor continuity; a beginning-reaching tail supersedes the prefix;
// everything else falls back to the full read within this same epoch (events
// keep buffering meanwhile).
const start = res.page?.before ?? null;
if (res.page !== undefined && start !== null && start === tailStartCursor) {
// Continuity proven: keep prefix and paging state exactly as they are.
} else if (res.page !== undefined && start === null) {
adoptTailPage(res.page); // tail covers everything: prefix dropped, provably complete
} else {
res = await deps.loadMessages();
if (disposed || currentEpoch !== epoch) return;
adoptTailPage(undefined); // full transcript: no prefix, no cursors
}
} else if (opts.page !== undefined) {
// Fresh tail baseline (initial load / retry): any previously-loaded prefix is
// superseded by the new window chain.
adoptTailPage(res.page);
} else {
adoptTailPage(undefined);
}
const { messages, live, serverNowMs } = res;
// Rebuild path: make the freshly-built model visible only now, atomically — the old model
// stayed on screen throughout the refetch above (see rebuild). Initial load / retry pass no
// freshModel and keep operating on the current model.
if (freshModel) model = freshModel;
const target = model;
// Windowed loads seed the stats accrued before the window, so header chips and
// per-turn cumulative rows equal a full load (see seedPriorStats).
if (res.page !== undefined) seedPriorStats(target.stats, res.page.prior);
pushMessages(target, messages, now(), serverNowMs ?? null);
const dedup = buildDedupIndex(messages, 100);
// Replay the buffer (events that arrived while fetching history), with dedup; while a
@@ -322,16 +477,79 @@ export function createStreamController(deps: StreamControllerDeps): StreamContro
}
};
/**
* Scroll-up backfill: fetch the window before the oldest loaded one and prepend it.
* The window's messages build a FRESH model (its own negative id base, priors seeded,
* finalizeHistory closing its last Task — complete by construction, since a newer
* window follows), whose items freeze into the prefix. Guarded to the live phase: a
* rebuild in flight owns the loading pipeline, and its epoch bump discards any
* backfill that raced it.
*/
const loadOlder = async (): Promise<void> => {
if (disposed || phase !== "live" || failed) return;
if (older.loading || !older.hasMore || nextBefore === null) return;
const currentEpoch = epoch;
older.loading = true;
older.error = null;
deps.onModelChange();
try {
const res = await deps.loadMessages({
kind: "before",
cursor: nextBefore,
limit: OLDER_UNITS,
});
if (disposed || currentEpoch !== epoch) return;
// A before-request against a server without windowing support would return the
// full transcript with no envelope; prepending that would duplicate history.
if (res.page === undefined) throw new Error("windowed history not supported");
prependCount += 1;
const m = createStreamModel(localDecisions);
m.nextItemId = -prependCount * PREPEND_ID_SPAN;
seedPriorStats(m.stats, res.page.prior);
pushMessages(m, res.messages, now(), null);
finalizeHistory(m);
prefixItems = [...m.items, ...prefixItems];
// Child sessions live in exactly one window (a spawn is contained in its Task),
// so the merge is disjoint; newer windows' entries win defensively on a clash.
const mergedSubagents = new Map(m.subagents);
for (const [sid, sub] of prefixSubagents) mergedSubagents.set(sid, sub);
prefixSubagents = mergedSubagents;
nextBefore = res.page.before ?? null;
older.hasMore = res.page.before !== undefined;
outlineOffset = res.page.earlierTurns;
} catch (e) {
if (disposed || currentEpoch !== epoch) return;
older.error = e instanceof Error ? e.message : String(e);
} finally {
if (!disposed && currentEpoch === epoch) {
older.loading = false;
deps.onModelChange();
}
}
};
return {
get model() {
return model;
},
get prefixItems(): readonly ChatItem[] {
return prefixItems;
},
get prefixSubagents(): ReadonlyMap<string, StreamModel> {
return prefixSubagents;
},
get outlineOffset() {
return outlineOffset;
},
get older(): OlderHistoryState {
return older;
},
get pendingApprovals(): ReadonlyMap<string, PendingApproval> {
return pending;
},
load: () => {
epoch += 1;
return load(epoch);
return load(epoch, undefined, { page: { kind: "tail", limit: TAIL_UNITS } });
},
retry: async () => {
if (disposed || !failed) return;
@@ -341,11 +559,15 @@ export function createStreamController(deps: StreamControllerDeps): StreamContro
// history straight onto it would duplicate the entire conversation (pushMessages appends with
// no id-dedup). load() swaps the fresh model in only once the refetch succeeds, then replays
// the still-accumulating buffer into it; localDecisions carry over via the shared set.
// The refetch is a fresh TAIL baseline: any retained prefix is superseded on success.
deps.onError(null);
deps.onLoading(true);
epoch += 1;
await load(epoch, createStreamModel(localDecisions));
await load(epoch, createStreamModel(localDecisions), {
page: { kind: "tail", limit: TAIL_UNITS },
});
},
loadOlder,
handleOmni: (msg, eventId = null) => {
if (disposed) return;
if (eventId !== null) lastEventId = eventId;
+27
View File
@@ -139,6 +139,33 @@ export function createTaskStatsTracker(): TaskStatsTracker {
};
}
/**
* Seed the tracker with the cumulative stats accrued BEFORE a partial history window
* (MessagesPageInfo.prior from a windowed GET /messages): finished-Task elapsed,
* subagent token totals, and the last session/context token readings. Applied before
* the window's messages replay, so header chips and per-turn cumulative rows land on
* the same figures a full-transcript load computes. sessionTotal/contextNow are only
* "last seen" fallbacks — any token_usage inside the window overwrites them; the
* elapsed/subagent/delta baselines genuinely accumulate on top.
*/
export function seedPriorStats(
t: TaskStatsTracker,
prior: {
subagentTokens: number;
elapsedMs: number;
sessionTokens: number;
contextTokens: number;
},
): void {
t.subagentTotal = prior.subagentTokens;
t.sessionElapsedMs = prior.elapsedMs;
t.sessionTotal = prior.sessionTokens;
t.contextNow = prior.contextTokens;
// The first in-window stats row's context delta measures against the pre-window
// occupancy, not against zero.
t.contextAtLastStats = prior.contextTokens;
}
/** This Task's bucketed accumulation (shared by main + subagent sessions; for real-time cost calc). */
function addBuckets(t: TaskStatsTracker, p: TokenUsagePayload): void {
t.taskCacheRead += p.request.cache_read;
+6
View File
@@ -758,6 +758,12 @@ Scenarios:
statusCompacting: "Compacting",
pendingApprovals: (n: number) => `${n} pending approval${n > 1 ? "s" : ""}`,
jumpToLatest: "Jump to latest",
/** Top-of-stream affordance while the previous history window is being fetched (scroll-up backfill). */
loadingEarlier: "Loading earlier messages…",
/** Top-of-stream affordance after a backfill failure: click to retry fetching the previous window. */
loadEarlierRetry: "Failed to load earlier messages — click to retry",
/** Top-of-stream marker once the loaded history reaches the very beginning (shown only after a backfill happened). */
historyBeginning: "Beginning of conversation",
/** Conversation minimap (tick rail over the stream's left gutter): rail aria-label. */
outlineTitle: "Outline",
/** Tick accessible name: turn number + the question (or the no-text placeholder). */
+6
View File
@@ -736,6 +736,12 @@ Benchmark:
statusCompacting: "压缩中",
pendingApprovals: (n: number) => `${n} 个待审批`,
jumpToLatest: "回到最新消息",
/** Top-of-stream affordance while the previous history window is being fetched (scroll-up backfill). */
loadingEarlier: "正在加载更早的对话…",
/** Top-of-stream affordance after a backfill failure: click to retry fetching the previous window. */
loadEarlierRetry: "更早的对话加载失败,点击重试",
/** Top-of-stream marker once the loaded history reaches the very beginning (shown only after a backfill happened). */
historyBeginning: "已是对话开头",
/** Conversation minimap (tick rail over the stream's left gutter): rail aria-label. */
outlineTitle: "对话索引",
/** Tick accessible name: turn number + the question (or the no-text placeholder). */