From 09c26ba17f47363670fbd403d8234fc03b7e15c4 Mon Sep 17 00:00:00 2001 From: zhangqingkun976 <1047045074@QQ.COM> Date: Tue, 22 Sep 2026 09:33:58 +0800 Subject: [PATCH] feat(agent-runtime): describe a failed compaction's range without a model MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rebased onto main (79cd7ae8); the rebase applied cleanly, no conflicts. When summary generation fails, the checkpoint is now built from the range it covers — a deterministic description plus a retained tail — instead of a notice that leaves the next window without anything to restore. Both degraded layers share one helper, so an empty tail can never be persisted for a completed turn. C:\Users\10470\.pi-desktop\scratch\cb9e2d41-8a55-4a18-899d-92ca2791ab51\commit-688-r2.txt --- apps/desktop/test/context-compaction.test.mjs | 19 +- .../src/compaction-trajectory.test.ts | 313 +++++++++++++++ .../src/compaction-trajectory.ts | 379 ++++++++++++++++++ packages/agent-runtime/src/runtime.test.ts | 87 +++- packages/agent-runtime/src/runtime.ts | 134 ++++++- 5 files changed, 900 insertions(+), 32 deletions(-) create mode 100644 packages/agent-runtime/src/compaction-trajectory.test.ts create mode 100644 packages/agent-runtime/src/compaction-trajectory.ts diff --git a/apps/desktop/test/context-compaction.test.mjs b/apps/desktop/test/context-compaction.test.mjs index c12d741c68..69928734e4 100644 --- a/apps/desktop/test/context-compaction.test.mjs +++ b/apps/desktop/test/context-compaction.test.mjs @@ -292,13 +292,22 @@ test("every compaction announces itself once, on top of the specific toasts", () }); test("a failed compaction checkpoint still restores a non-empty context", () => { - // A retained-tail fallback must not persist an empty tail on a completed - // turn: with no real summary to carry the boundary, an empty tail restores - // as an empty context after a runtime rebuild (model switch / restart) — - // the session reads as if it had just started (#224). + // A degraded checkpoint must not persist an empty tail on a completed turn: + // with no real summary to carry the boundary, an empty tail restores as an + // empty context after a runtime rebuild (model switch / restart) — the + // session reads as if it had just started (#224). + // + // Both degraded layers (the deterministic record and the retained-tail + // notice) share one selection, so the rule cannot hold in one path and be + // forgotten in the other. assert.match( runtime, - /const retainedTail =\s*preparation\.retainedTail\.length > 0\s*\? preparation\.retainedTail\s*: selectRetainedUserMessages\(/, + /private retainedTailForDegradedCheckpoint\(\s*preparation: ShapedPreparation,\s*\): AgentMessage\[\] \{\s*return preparation\.retainedTail\.length > 0\s*\? preparation\.retainedTail\s*: selectRetainedUserMessages\(/, + ); + assert.equal( + (runtime.match(/retainedTailForDegradedCheckpoint\(preparation\)/g) ?? []) + .length, + 2, ); assert.match( runtime, diff --git a/packages/agent-runtime/src/compaction-trajectory.test.ts b/packages/agent-runtime/src/compaction-trajectory.test.ts new file mode 100644 index 0000000000..014980121a --- /dev/null +++ b/packages/agent-runtime/src/compaction-trajectory.test.ts @@ -0,0 +1,313 @@ +import { describe, expect, it } from "vitest"; +import { + TRAJECTORY_MAX_GOAL_CHARS, + TRAJECTORY_MAX_NEXT_STEPS, + TRAJECTORY_MAX_OPEN_ITEMS, + buildTrajectorySummary, + extractGoal, + extractNextSteps, + extractOpenItems, + type TrajectoryMessage, +} from "./compaction-trajectory.js"; + +/** + * The property that makes a trajectory checkpoint a real summary rather than a + * note: the same minimum length and section headings the runtime's own summary + * check enforces, so the next summarization request can carry it forward like + * any other summary instead of being a special case. + */ +const REQUIRED_SECTIONS = ["## Goal", "## Progress", "## Next Steps"] as const; + +function carryableAsSummary(text: string): boolean { + if (text.trim().length < 200) return false; + return REQUIRED_SECTIONS.every((section) => + new RegExp(`^${section}\\s*$`, "m").test(text), + ); +} + +function user(text: string): TrajectoryMessage { + return { role: "user", content: text }; +} + +function assistant( + text: string, + calls: Array<{ id?: string; name: string; arguments?: unknown }> = [], +): TrajectoryMessage { + return { + role: "assistant", + content: [ + ...(text ? [{ type: "text", text }] : []), + ...calls.map((call) => ({ + type: "toolCall", + id: call.id, + name: call.name, + arguments: call.arguments ?? {}, + })), + ], + }; +} + +function failure( + toolName: string, + text: string, + toolCallId?: string, +): TrajectoryMessage { + return { + role: "toolResult", + toolName, + isError: true, + toolCallId, + content: [{ type: "text", text }], + }; +} + +describe("extractGoal", () => { + it("takes the last user request verbatim", () => { + expect( + extractGoal([user("first ask"), user("second ask")]), + ).toBe("second ask"); + }); + + it("ignores an empty or non-text user message", () => { + expect(extractGoal([user("real ask"), user(" ")])).toBe("real ask"); + expect(extractGoal([{ role: "user" }])).toBeUndefined(); + }); + + it("bounds a goal that would dominate the checkpoint", () => { + const goal = extractGoal([user("x".repeat(TRAJECTORY_MAX_GOAL_CHARS + 50))]); + expect(goal?.length).toBe(TRAJECTORY_MAX_GOAL_CHARS + 1); + expect(goal?.endsWith("…")).toBe(true); + }); + + it("returns undefined when the range holds no user request", () => { + expect(extractGoal([assistant("done")])).toBeUndefined(); + }); +}); + +describe("extractOpenItems", () => { + it("names a failure nothing repaired", () => { + expect( + extractOpenItems([ + assistant("", [{ id: "call-1", name: "Bash", arguments: { command: "x" } }]), + failure("Bash", "command failed: exit 1", "call-1"), + ]), + ).toEqual(["Bash: command failed: exit 1"]); + }); + + it("drops a failure a later edit to the same file repaired", () => { + expect( + extractOpenItems([ + assistant("", [{ id: "read-1", name: "Read", arguments: { file_path: "a.ts" } }]), + failure("Read", "file not found", "read-1"), + assistant("", [{ id: "write-1", name: "Write", arguments: { file_path: "a.ts" } }]), + // The repair has to have landed: a Write that never returned, or one + // that failed, is not evidence that the file became readable. + { + role: "toolResult", + toolName: "Write", + isError: false, + toolCallId: "write-1", + content: [{ type: "text", text: "written" }], + }, + ]), + ).toEqual([]); + }); + + it("keeps a failure whose repair came before it", () => { + // The edit belongs to earlier work; the failure happened afterwards and is + // still unexplained, which is exactly what a next window needs to know. + expect( + extractOpenItems([ + assistant("", [{ id: "write-1", name: "Write", arguments: { file_path: "a.ts" } }]), + assistant("", [{ id: "read-1", name: "Read", arguments: { file_path: "a.ts" } }]), + failure("Read", "file not found", "read-1"), + ]), + ).toEqual(["Read: file not found"]); + }); + + it("keeps a failure it cannot attribute to a path", () => { + expect( + extractOpenItems([failure("Bash", "npm test failed", "unknown-call")]), + ).toEqual(["Bash: npm test failed"]); + }); + + it("takes only the first line and bounds the list", () => { + const items = extractOpenItems([ + failure("Bash", "first line\nsecond line"), + ...Array.from({ length: TRAJECTORY_MAX_OPEN_ITEMS + 3 }, (_, index) => + failure("Bash", `failure ${index}`), + ), + ]); + expect(items[0]).toBe("Bash: first line"); + expect(items).toHaveLength(TRAJECTORY_MAX_OPEN_ITEMS); + }); + + it("does not repeat an identical failure", () => { + expect( + extractOpenItems([failure("Bash", "same"), failure("Bash", "same")]), + ).toEqual(["Bash: same"]); + }); + + it("ignores a result that succeeded", () => { + expect( + extractOpenItems([ + { role: "toolResult", toolName: "Bash", isError: false, content: "ok" }, + ]), + ).toEqual([]); + }); + + it("matches a result to the nearest preceding call, never to a later one", () => { + // OpenAI-compatible local servers emit ids like `1`, `2` per response, so a + // whole range contains repeats. The call a failure belongs to is the one + // that precedes it; a later call reusing the id says nothing about it. + expect( + extractOpenItems([ + assistant("", [{ id: "1", name: "Read", arguments: { file_path: "a.ts" } }]), + failure("Read", "file not found", "1"), + assistant("", [{ id: "1", name: "Write", arguments: { file_path: "a.ts" } }]), + { + role: "toolResult", + toolName: "Write", + isError: false, + toolCallId: "1", + content: [{ type: "text", text: "written" }], + }, + ]), + ).toEqual(["Read: file not found"]); + }); + + it("does not treat a failed edit as a repair", () => { + expect( + extractOpenItems([ + assistant("", [{ id: "r1", name: "Read", arguments: { file_path: "a.ts" } }]), + failure("Read", "file not found", "r1"), + assistant("", [{ id: "w1", name: "Write", arguments: { file_path: "a.ts" } }]), + failure("Write", "the write itself failed", "w1"), + ]), + ).toEqual(["Read: file not found", "Write: the write itself failed"]); + }); + + it("does not let a call that never returned hide a failure", () => { + expect( + extractOpenItems([ + assistant("", [{ id: "r1", name: "Read", arguments: { file_path: "a.ts" } }]), + failure("Read", "file not found", "r1"), + // Unverifiable, so it cannot count as the repair that resolved the + // failure above: an unresolved call is not evidence of anything. + assistant("", [{ id: "w1", name: "Write", arguments: { file_path: "a.ts" } }]), + ]), + ).toEqual(["Read: file not found"]); + }); +}); + +describe("extractNextSteps", () => { + it("reads unchecked checkboxes and literal TODOs", () => { + expect( + extractNextSteps([ + assistant("- [ ] wire the new layer\n- [x] done\nTODO: run the suite"), + ]), + ).toEqual(["wire the new layer", "run the suite"]); + }); + + it("does not turn prose into a task list", () => { + expect( + extractNextSteps([ + assistant("## Next Steps\n\nEverything below was finished already."), + ]), + ).toEqual([]); + }); + + it("reads the last assistant message that has text", () => { + expect( + extractNextSteps([assistant("TODO: old"), assistant(""), assistant("TODO: new")]), + ).toEqual(["new"]); + }); + + it("bounds the list", () => { + const steps = extractNextSteps([ + assistant( + Array.from({ length: TRAJECTORY_MAX_NEXT_STEPS + 4 }, (_, i) => `- [ ] step ${i}`).join("\n"), + ), + ]); + expect(steps).toHaveLength(TRAJECTORY_MAX_NEXT_STEPS); + }); +}); + +describe("buildTrajectorySummary", () => { + const range: TrajectoryMessage[] = [ + user("older ask"), + assistant("", [{ id: "call-1", name: "Bash", arguments: { command: "ls" } }]), + failure("Bash", "cannot ls here", "call-1"), + user("the ask that matters"), + assistant("TODO: finish the record"), + ]; + + it("carries the goal, the unresolved failures and the next step", () => { + const summary = buildTrajectorySummary({ messages: range }); + expect(summary.sections).toContain("## Goal"); + expect(summary.sections).toContain("the ask that matters"); + expect(summary.sections).toContain("## Blocked"); + expect(summary.sections).toContain("- Bash: cannot ls here"); + expect(summary.sections).toContain("## Next Steps"); + expect(summary.sections).toContain("- finish the record"); + expect(summary.stats).toEqual({ + messages: 5, + toolCalls: 1, + failedToolCalls: 1, + }); + expect(summary.goal).toBe("the ask that matters"); + }); + + it("carries as a real summary, not a note", () => { + // The headings and the minimum length are the ones the runtime's summary + // check enforces, so a trajectory checkpoint can be carried into the next + // summarization request like any other summary instead of being a special + // case. + const summary = buildTrajectorySummary({ messages: range }); + expect(carryableAsSummary(summary.sections)).toBe(true); + }); + + it("says so when a section has nothing to carry", () => { + const summary = buildTrajectorySummary({ messages: [assistant("done")] }); + expect(summary.sections).toContain("(no user request was recorded in this range)"); + expect(summary.sections).toContain("## Blocked\n(none recorded)"); + expect(summary.sections).toContain("## Next Steps\n(no open next step was recorded)"); + expect(summary.goal).toBeUndefined(); + }); + + it("keeps the carried summary ahead of the sections, bounded", () => { + const summary = buildTrajectorySummary({ + messages: range, + previousSummary: " The earlier task summary. ", + }); + expect(summary.carried).toBe("The earlier task summary."); + expect(summary.summary.startsWith("The earlier task summary.")).toBe(true); + expect(summary.sections).not.toContain("The earlier task summary."); + }); + + it("bounds a carried summary that has grown past its budget", () => { + const summary = buildTrajectorySummary({ + messages: range, + previousSummary: "x".repeat(10_000), + maxPreviousChars: 64, + }); + expect(summary.carried?.length).toBe(65); + }); + + it("treats a blank carried summary as none", () => { + expect(buildTrajectorySummary({ messages: range, previousSummary: " " }).carried) + .toBeUndefined(); + }); + + it("never throws on messages it does not understand", () => { + const summary = buildTrajectorySummary({ + messages: [ + { role: "custom" }, + { role: "assistant", content: "not a block list" }, + { role: "toolResult", isError: true }, + ], + }); + expect(summary.sections).toContain("## Goal"); + expect(summary.openItems).toEqual(["tool"]); + }); +}); diff --git a/packages/agent-runtime/src/compaction-trajectory.ts b/packages/agent-runtime/src/compaction-trajectory.ts new file mode 100644 index 0000000000..59981d5edc --- /dev/null +++ b/packages/agent-runtime/src/compaction-trajectory.ts @@ -0,0 +1,379 @@ +/** + * Deterministic compaction recovery. + * + * Why this exists + * --------------- + * When the summary request fails, the runtime falls back to a carried-forward + * summary plus a recovery notice: real history is dropped and the next window is + * told nothing about what the work was. The failure this module was written for + * is the evidence: 750k tokens of work replaced by a notice plus a + * machine-generated ledger of file names, with no statement of the goal, of what + * went wrong, or of what was left to do. + * + * The layer this adds is model-free, so it cannot fail on a provider and it + * costs no request: the range is described *mechanically* — the last request + * that was being worked on, verbatim; how much work the range holds and how much + * of it failed; which failures were never resolved; and the open items the last + * assistant message stated. It slots between a failed model summary and the + * retained-tail notice. + * + * What this file is not + * --------------------- + * It does not talk to a model, read the filesystem, or know about checkpoints. + * Everything here is a pure function of the messages it is given, which is what + * lets the whole degradation ladder be tested without a provider stub. The + * narrative a model would have written is the runtime's business; this is the + * part that must be true even when no model answers. + * + * Duplication with the session ledger is deliberate but bounded: files, + * commands and range totals are the ledger's job, so a trajectory summary never + * restates them. What it adds is exactly what the ledger lacks — the goal, the + * unresolved failures, and the next step. + */ + +/** The verbatim goal is the one thing a next window cannot reconstruct. */ +export const TRAJECTORY_MAX_GOAL_CHARS = 1_500; + +/** Unresolved failures worth naming; past this they stop being actionable. */ +export const TRAJECTORY_MAX_OPEN_ITEMS = 8; +export const TRAJECTORY_OPEN_ITEM_CHARS = 200; + +/** Open items the last assistant message stated. */ +export const TRAJECTORY_MAX_NEXT_STEPS = 6; + +/** Carried-forward summary budget, before the range's own description. */ +export const TRAJECTORY_MAX_PREVIOUS_CHARS = 4_000; + +/** + * The message fields this module reads. Structural rather than pi's message + * union so a test can pass a literal, and so a future message role this module + * does not understand simply contributes nothing instead of failing a cast. + */ +export type TrajectoryMessage = { + role: string; + content?: unknown; + isError?: boolean; + toolName?: string; + toolCallId?: string; +}; + +export type TrajectoryStats = { + messages: number; + toolCalls: number; + failedToolCalls: number; +}; + +export type TrajectorySummary = { + /** + * The real summary a chained compaction was carrying, bounded. The runtime + * places it ahead of its recovery marker, so `stripCompactionFallbackNotice` + * recovers exactly this text for the next attempt — the sections below + * describe *this* range and must not be carried as if they were history. + */ + carried: string | undefined; + /** The mechanical description of this range; the runtime places it after the marker. */ + sections: string; + /** `carried` + `sections`, for a reader that has no marker to place. */ + summary: string; + stats: TrajectoryStats; + /** Named for the checkpoint's diagnostics; undefined when nothing recorded it. */ + goal: string | undefined; + openItems: string[]; + nextSteps: string[]; +}; + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +/** Text of every text block, in order. A string content is its own text. */ +function messageText(message: TrajectoryMessage): string { + const content = message.content; + if (typeof content === "string") return content; + if (!Array.isArray(content)) return ""; + const parts: string[] = []; + for (const block of content) { + if (isRecord(block) && block.type === "text" && typeof block.text === "string") { + parts.push(block.text); + } + } + return parts.join("\n"); +} + +function firstLine(text: string, maxChars: number): string { + const line = text + .split(/\r?\n/) + .map((candidate) => candidate.trim()) + .find((candidate) => candidate.length > 0); + if (!line) return ""; + return line.length > maxChars ? `${line.slice(0, maxChars)}…` : line; +} + +function pathArgument(args: Record): string | undefined { + for (const key of ["path", "file_path", "filePath"]) { + const value = args[key]; + if (typeof value === "string" && value.length > 0) return value; + } + return undefined; +} + +/** Tool-call blocks of one assistant message, with the arguments we can read. */ +function toolCallsOf( + message: TrajectoryMessage, +): Array<{ id: string | undefined; name: string; path: string | undefined }> { + const content = message.content; + if (!Array.isArray(content)) return []; + const calls: Array<{ + id: string | undefined; + name: string; + path: string | undefined; + }> = []; + for (const block of content) { + if (!isRecord(block) || block.type !== "toolCall") continue; + const name = typeof block.name === "string" ? block.name : ""; + if (!name) continue; + const args = isRecord(block.arguments) ? block.arguments : {}; + calls.push({ + id: typeof block.id === "string" ? block.id : undefined, + name, + path: pathArgument(args), + }); + } + return calls; +} + +/** + * The last request the range was working on, verbatim. + * + * Verbatim matters: Roo Code's prompt asks a summarizer for the same thing + * because a paraphrase is what lets a next window redo work whose wording it + * cannot tell apart from a different request. + */ +export function extractGoal( + messages: readonly TrajectoryMessage[], +): string | undefined { + for (let index = messages.length - 1; index >= 0; index--) { + const message = messages[index]; + if (message?.role !== "user") continue; + const text = messageText(message).trim(); + if (text.length === 0) continue; + return text.length > TRAJECTORY_MAX_GOAL_CHARS + ? `${text.slice(0, TRAJECTORY_MAX_GOAL_CHARS)}…` + : text; + } + return undefined; +} + +/** + * Failures in the range that nothing later repaired. + * + * "Unresolved" is the useful half: a test run that failed and was fixed by the + * next edit is not something a next window should act on, and listing it would + * teach the reader to skim this section. A failure whose replacement edit came + * later in the same range is therefore dropped; everything else — including a + * failure this module cannot attribute to a path — is kept, because an + * unexplained failure is exactly what a next window needs to know about. + * + * Two attribution rules keep that from hiding a real failure: + * + * - A call id is **not** unique across a range — OpenAI-compatible local + * servers emit `1`, `2`, … per response — so a result is matched to the + * nearest *preceding* call with that id, never to a later one. + * - A repair counts only when its own result came back without an error. A + * `Write` that failed repaired nothing, and a call that never returned is + * not evidence either, so neither suppresses an item. + */ +export function extractOpenItems( + messages: readonly TrajectoryMessage[], +): string[] { + const calls = new Map< + string, + Array<{ index: number; name: string; path: string | undefined }> + >(); + const resultIds = new Set(); + const failedIds = new Set(); + for (let index = 0; index < messages.length; index++) { + const message = messages[index]; + if (!message) continue; + if (message.role === "assistant") { + for (const call of toolCallsOf(message)) { + if (!call.id) continue; + const list = calls.get(call.id) ?? []; + list.push({ index, name: call.name, path: call.path }); + calls.set(call.id, list); + } + continue; + } + if (message.role !== "toolResult") continue; + if (typeof message.toolCallId !== "string") continue; + resultIds.add(message.toolCallId); + if (message.isError === true) failedIds.add(message.toolCallId); + } + + /** The call a result belongs to: the last one with that id before it. */ + const callBefore = (id: string, index: number) => { + const list = calls.get(id); + if (!list) return undefined; + for (let position = list.length - 1; position >= 0; position--) { + const call = list[position]; + if (call && call.index < index) return call; + } + return undefined; + }; + const repaired = (id: string | undefined, index: number): boolean => { + if (!id || !resultIds.has(id) || failedIds.has(id)) return false; + const call = callBefore(id, index); + return call?.index !== undefined; + }; + + // Paths an edit actually landed on, so a failure for that path is resolved. + const lastRepair = new Map(); + for (const [id, list] of calls) { + if (!repaired(id, Number.POSITIVE_INFINITY)) continue; + const call = list[list.length - 1]; + if (!call?.path) continue; + if (call.name !== "Write" && call.name !== "Edit" && call.name !== "MultiEdit") { + continue; + } + lastRepair.set(call.path, call.index); + } + + const items: string[] = []; + const seen = new Set(); + for (let index = 0; index < messages.length; index++) { + const message = messages[index]; + if (!message || message.role !== "toolResult" || message.isError !== true) { + continue; + } + const call = + typeof message.toolCallId === "string" + ? callBefore(message.toolCallId, index) + : undefined; + if (call?.path) { + const repair = lastRepair.get(call.path); + if (repair !== undefined && repair > index) continue; + } + const name = + call?.name || + (typeof message.toolName === "string" ? message.toolName : "tool"); + const detail = firstLine(messageText(message), TRAJECTORY_OPEN_ITEM_CHARS); + const item = detail ? `${name}: ${detail}` : name; + if (seen.has(item)) continue; + seen.add(item); + items.push(item); + if (items.length >= TRAJECTORY_MAX_OPEN_ITEMS) break; + } + return items; +} + + +/** + * Open items the last assistant message stated. + * + * Only markers that cannot appear by accident are read — an unchecked checkbox + * and a literal `TODO`. A looser rule (any line under a "next steps" heading) + * would turn prose into a task list the model never committed to, and the + * sections above already carry the goal verbatim. + */ +export function extractNextSteps( + messages: readonly TrajectoryMessage[], +): string[] { + for (let index = messages.length - 1; index >= 0; index--) { + const message = messages[index]; + if (message?.role !== "assistant") continue; + const text = messageText(message); + if (text.trim().length === 0) continue; + const steps: string[] = []; + for (const rawLine of text.split(/\r?\n/)) { + const checkbox = rawLine.match(/^\s*[-*]\s*\[\s*\]\s*(.+)$/); + const todo = rawLine.match(/^\s*(?:[-*]\s*)?TODO\b[:\s]*(.*)$/i); + const value = (checkbox?.[1] ?? todo?.[1] ?? "").trim(); + if (!value) continue; + steps.push( + value.length > TRAJECTORY_OPEN_ITEM_CHARS + ? `${value.slice(0, TRAJECTORY_OPEN_ITEM_CHARS)}…` + : value, + ); + if (steps.length >= TRAJECTORY_MAX_NEXT_STEPS) break; + } + return steps; + } + return []; +} + +const TRAJECTORY_GOAL_EMPTY = "(no user request was recorded in this range)"; +const TRAJECTORY_BLOCKED_EMPTY = "(none recorded)"; +const TRAJECTORY_NEXT_EMPTY = "(no open next step was recorded)"; + +/** + * The mechanical replacement for a model summary. + * + * The section headings are the ones the A check enforces, on purpose: a + * trajectory checkpoint is then structurally a summary, so the next + * summarization request can carry it forward and every reader — human or model — + * sees the same shape regardless of which layer produced it. + * + * `previousSummary` is placed ahead of the caller's marker, exactly like the + * retained-tail fallback does, so `stripCompactionFallbackNotice` can recover + * the real carried-forward history on a chained compaction and this module's + * description of the *previous* range is dropped rather than cemented. + */ +export function buildTrajectorySummary(input: { + messages: readonly TrajectoryMessage[]; + previousSummary?: string; + /** Bound for the carried-forward text; the sections bound themselves. */ + maxPreviousChars?: number; +}): TrajectorySummary { + const messages = input.messages; + const stats: TrajectoryStats = { + messages: messages.length, + toolCalls: 0, + failedToolCalls: 0, + }; + for (const message of messages) { + if (message.role === "assistant") { + stats.toolCalls += toolCallsOf(message).length; + } else if (message.role === "toolResult" && message.isError === true) { + stats.failedToolCalls += 1; + } + } + + const goal = extractGoal(messages); + const openItems = extractOpenItems(messages); + const nextSteps = extractNextSteps(messages); + const previousBudget = input.maxPreviousChars ?? TRAJECTORY_MAX_PREVIOUS_CHARS; + const previous = + typeof input.previousSummary === "string" && input.previousSummary.trim().length > 0 + ? input.previousSummary.trim().length > previousBudget + ? `${input.previousSummary.trim().slice(0, previousBudget)}…` + : input.previousSummary.trim() + : undefined; + + const sections = [ + "## Goal", + goal ?? TRAJECTORY_GOAL_EMPTY, + "", + "## Progress", + `${stats.messages} messages and ${stats.toolCalls} tool calls were summarized without a model pass; ${stats.failedToolCalls} tool call(s) in the range failed. Files, commands and totals for this range are listed in the session ledger below.`, + "", + "## Blocked", + openItems.length > 0 + ? openItems.map((item) => `- ${item}`).join("\n") + : TRAJECTORY_BLOCKED_EMPTY, + "", + "## Next Steps", + nextSteps.length > 0 + ? nextSteps.map((item) => `- ${item}`).join("\n") + : TRAJECTORY_NEXT_EMPTY, + ].join("\n"); + + return { + carried: previous, + sections, + summary: previous ? `${previous}\n\n${sections}` : sections, + stats, + goal, + openItems, + nextSteps, + }; +} diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index cc0346e2bb..4e7342b32d 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -5340,7 +5340,7 @@ describe("DesktopAgentRuntime per-turn context protection", () => { await runtime.dispose(); }); - it("persists a retained-tail fallback when automatic summary generation fails", async () => { + it("persists a deterministic record when automatic summary generation fails", async () => { const onEvent = vi.fn(); const host = { call: vi.fn().mockResolvedValue(undefined) }; const runtime = createRuntime({ @@ -5378,16 +5378,23 @@ describe("DesktopAgentRuntime per-turn context protection", () => { const checkpoint = host.call.mock.calls[0]?.[0] === "session.appendCompaction" ? (host.call.mock.calls[0]?.[1] as any).compaction : undefined; - expect(checkpoint).toEqual( + // The deterministic record is the first rung: the model summary failed, so + // the durable checkpoint carries what no later request can reconstruct — + // the goal verbatim, the failures nothing repaired, the next step — instead + // of only a notice. `strategy` keeps it readable as a real summary. + expect(host.call).toHaveBeenCalledTimes(1); + expect(checkpoint.throughMessageId).toBe("recent-user"); + expect(checkpoint.details).toEqual( expect.objectContaining({ - throughMessageId: "recent-user", - details: expect.objectContaining({ - fallback: "retained_tail", - failureCode: "CONTEXT_COMPACTION_FAILED", - retainedTailMode: "completed_turn", - }), + strategy: "trajectory", + failureCode: "CONTEXT_COMPACTION_FAILED", + retainedTailMode: "completed_turn", + trajectory: expect.objectContaining({ messages: 2, goal: true }), }), ); + expect(checkpoint.summary).toContain("continue the task"); + expect(checkpoint.summary).toContain("## Goal"); + expect(checkpoint.summary).toContain(COMPACTION_FALLBACK_MARKER); expect((runtime as any).fullEntries).toHaveLength(2); expect((runtime as any).agent.state.messages.filter((message: any) => message.role !== "system")[0]).toEqual( expect.objectContaining({ role: "compactionSummary" }), @@ -5407,6 +5414,57 @@ describe("DesktopAgentRuntime per-turn context protection", () => { await runtime.dispose(); }); + it("still writes the retained-tail notice when the record cannot fit", async () => { + // The deterministic record is tried first because it is free, but it is not + // the last resort: when it does not fit the safe budget the notice below it + // still runs, so a summary failure never ends without a checkpoint. + const host = { call: vi.fn().mockResolvedValue(undefined) }; + const runtime = createRuntime({ + host, + history: [ + { + id: "old-user", + role: "user", + content: "older task context", + createdAt: "2026-07-28T00:00:00Z", + status: "complete", + }, + { + id: "recent-user", + role: "user", + content: "continue the task", + createdAt: "2026-07-28T00:00:01Z", + status: "complete", + }, + ], + }); + vi.spyOn(runtime as any, "generateCompaction").mockResolvedValue({ + ok: false, + error: { + code: "summarization_failed", + message: "provider terminated the summary request", + }, + }); + const persist = vi.spyOn(runtime as any, "persistCheckpoint"); + persist.mockResolvedValueOnce("oversized"); + persist.mockResolvedValue("persisted"); + + await expect( + (runtime as any).runCompaction("threshold", false), + ).resolves.toBe(true); + + expect(persist).toHaveBeenCalledTimes(2); + const first = persist.mock.calls[0]?.[0] as any; + const second = persist.mock.calls[1]?.[0] as any; + expect(first.details.strategy).toBe("trajectory"); + // The last resort keeps its own identity rather than being a copy of the + // record that could not fit. + expect(second.details.fallback).toBe("retained_tail"); + expect(second.details.strategy).toBeUndefined(); + expect(second.summary).not.toContain("## Goal"); + await runtime.dispose(); + }); + it("shrinks a terminal checkpoint tail when a new prompt leaves no history to summarize", async () => { const host = { call: vi.fn().mockResolvedValue(undefined) }; const runtime = createRuntime({ @@ -5447,16 +5505,21 @@ describe("DesktopAgentRuntime per-turn context protection", () => { true, ); + // No new history to summarize, so the carried summary is the only older + // context there is: the durable checkpoint must carry it forward intact and + // the deterministic record must describe this range on top of it. expect(generateCompaction).not.toHaveBeenCalled(); expect(host.call).toHaveBeenCalledWith( "session.appendCompaction", expect.objectContaining({ compaction: expect.objectContaining({ - details: expect.objectContaining({ fallback: "retained_tail" }), + details: expect.objectContaining({ strategy: "trajectory" }), summary: expect.stringContaining("The previous task summary."), }), }), ); + const appended = (host.call.mock.calls[0]?.[1] as any).compaction; + expect(appended.summary).toContain("## Goal"); await runtime.dispose(); }); @@ -5568,11 +5631,15 @@ describe("DesktopAgentRuntime per-turn context protection", () => { "session.appendCompaction", expect.objectContaining({ compaction: expect.objectContaining({ - details: expect.objectContaining({ fallback: "retained_tail" }), + details: expect.objectContaining({ strategy: "trajectory" }), summary: expect.stringContaining("The earlier task summary."), }), }), ); + // The carried history is recovered from the notice marker, so a chained + // fallback does not cement this range's description as history (#224). + const appended = (host.call.mock.calls[0]?.[1] as any).compaction; + expect(appended.summary).toContain("## Goal"); await runtime.dispose(); }); diff --git a/packages/agent-runtime/src/runtime.ts b/packages/agent-runtime/src/runtime.ts index 1877654acd..deb55ec755 100644 --- a/packages/agent-runtime/src/runtime.ts +++ b/packages/agent-runtime/src/runtime.ts @@ -200,6 +200,10 @@ import { estimateSummaryPromptTokens, reduceSummaryInput, } from "./compaction-summary-input.js"; +import { + buildTrajectorySummary, + TRAJECTORY_MAX_PREVIOUS_CHARS, +} from "./compaction-trajectory.js"; import { mergeProviderHeaders, providerHeadersEqual, @@ -6260,19 +6264,10 @@ Delegation rules: // — model switch, restart — the session restores as if it had just started. // Fall back to the newest user messages under the same budget so the // failure path still restores a bounded, non-empty context (#224). - const retainedTail = - preparation.retainedTail.length > 0 - ? preparation.retainedTail - : selectRetainedUserMessages( - preparation.messagesToSummarize.filter( - (message): message is UserMessage => message.role === "user", - ), - preparation.settings.keepRecentTokens, - ); return this.createCheckpoint( { ...preparation, - retainedTail, + retainedTail: this.retainedTailForDegradedCheckpoint(preparation), }, throughMessageId, summary, @@ -6287,6 +6282,89 @@ Delegation rules: ); } + /** + * The deterministic layer between a failed model summary and the + * retained-tail notice. It costs no request and cannot fail on a provider, + * so trying it first is free; when it does not fit the safe budget either, + * the notice above still runs and stays the last resort. + */ + private createTrajectoryCheckpoint( + preparation: ShapedPreparation, + throughMessageId: string, + maxSummaryChars: number, + retentionMode: CompactionRetentionMode, + failureMessage: string, + ): ContextCompactionRecord { + const trajectory = buildTrajectorySummary({ + messages: [ + ...preparation.messagesToSummarize, + ...preparation.turnPrefixMessages, + ], + previousSummary: preparation.previousSummary, + maxPreviousChars: Math.min(TRAJECTORY_MAX_PREVIOUS_CHARS, maxSummaryChars), + }); + const continuation = + retentionMode === "active_turn" + ? "The provider is continuing the active turn. Use the one retained latest user request as the source of truth for that continuation." + : "The previous turn is complete. Treat this summary as historical context; the next user message is the only new task to execute."; + // Same layout as the retained-tail fallback: the carried summary first, the + // marker, then what this layer has to say, so a chained compaction recovers + // exactly the carried history and drops this range's description instead of + // cementing it (#224). + const summary = [ + trajectory.carried ?? COMPACTION_FALLBACK_NO_SUMMARY, + COMPACTION_FALLBACK_MARKER, + "The automatic summary request did not complete. Older messages before this checkpoint are omitted from the next model request.", + `${continuation} The deterministic record below describes what they contained.`, + trajectory.sections, + ].join("\n\n"); + return this.createCheckpoint( + { + ...preparation, + retainedTail: this.retainedTailForDegradedCheckpoint(preparation), + }, + throughMessageId, + summary, + undefined, + { + ...this.checkpointDetails(preparation), + ...this.retainedReasoningForCheckpoint(preparation), + // `trajectory` is an outcome of the degradation chain, not a policy a + // user can select, so it stays out of `CompactionStrategy` — the + // shared reader already treats anything but `fresh_window` as + // summarized, and this is a real summary, just not a model's. + strategy: "trajectory", + retainedTailMode: retentionMode, + failureCode: "CONTEXT_COMPACTION_FAILED", + failureMessage: boundedText( + failureMessage, + COMPACTION_FALLBACK_MAX_SUMMARY_CHARS, + ), + trajectory: { + messages: trajectory.stats.messages, + toolCalls: trajectory.stats.toolCalls, + failedToolCalls: trajectory.stats.failedToolCalls, + openItems: trajectory.openItems.length, + nextSteps: trajectory.nextSteps.length, + goal: trajectory.goal !== undefined, + }, + }, + ); + } + + private retainedTailForDegradedCheckpoint( + preparation: ShapedPreparation, + ): AgentMessage[] { + return preparation.retainedTail.length > 0 + ? preparation.retainedTail + : selectRetainedUserMessages( + preparation.messagesToSummarize.filter( + (message): message is UserMessage => message.role === "user", + ), + preparation.settings.keepRecentTokens, + ); + } + /** * The recovery path retains less than a normal checkpoint: its summary is a * carried-forward one rather than a fresh one, so the retained messages are @@ -6476,17 +6554,39 @@ Delegation rules: ); return false; } + const maxSummaryChars = Math.max( + 256, + Math.min( + COMPACTION_FALLBACK_MAX_SUMMARY_CHARS, + Math.floor(budget.hardLimit * 0.75), + ), + ); + // The deterministic record gets its turn before the retained-tail notice: + // it costs no request, so trying it first is free, and it carries what a + // next window cannot reconstruct (the goal verbatim, the failures nothing + // repaired, the next step the last assistant message stated) even though + // every model attempt failed. When it does not fit the safe budget either, + // the notice below still runs and stays the last resort. + const degraded = this.createTrajectoryCheckpoint( + fallbackPreparation, + throughMessageId, + maxSummaryChars, + retentionMode, + failureMessage, + ); + const degradedPersisted = await this.persistCheckpoint( + degraded, + reason, + willRetry, + true, + "retained_tail", + ); + if (degradedPersisted === "persisted") return true; const checkpoint = this.createFallbackCheckpoint( fallbackPreparation, throughMessageId, - Math.max( - 256, - Math.min( - COMPACTION_FALLBACK_MAX_SUMMARY_CHARS, - Math.floor(budget.hardLimit * 0.75), - ), - ), + maxSummaryChars, retentionMode, ); const persisted = await this.persistCheckpoint(