From a90bcb18c64c918e004268a220aadf070cf50be3 Mon Sep 17 00:00:00 2001 From: zhangqingkun976 <1047045074@QQ.COM> Date: Tue, 22 Sep 2026 09:34:23 +0800 Subject: [PATCH] feat(agent-runtime): tier old tool results when the outgoing view is under pressure Rebased onto main (79cd7ae8). One conflict, from #733 extracting the budget formula: `ContextBudget` and the `retainedUserMessageBudget` doc comment now live in `context-budget.ts`, so they stay out of `runtime.ts`, and `narrowToolResultsUnderPressure` sits beside the import. Standalone shape: the gate is the fixed `TOOL_RESULT_TIER_PRESSURE` share of the hard limit. The per-model configurable threshold is in #721, which contains this change. C:\Users\10470\.pi-desktop\scratch\cb9e2d41-8a55-4a18-899d-92ca2791ab51\commit-705-r2.txt --- packages/agent-runtime/src/runtime.test.ts | 100 ++++ packages/agent-runtime/src/runtime.ts | 40 +- .../src/tool-result-tier.test.ts | 402 ++++++++++++++ .../agent-runtime/src/tool-result-tier.ts | 508 ++++++++++++++++++ 4 files changed, 1048 insertions(+), 2 deletions(-) create mode 100644 packages/agent-runtime/src/tool-result-tier.test.ts create mode 100644 packages/agent-runtime/src/tool-result-tier.ts diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index cc0346e2bb..f5ec42514e 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -8703,3 +8703,103 @@ describe("context estimate calibration", () => { }); }); }); + +describe("tool-result tiering under context pressure", () => { + /** A stored Read window the way host-core renders one: header, then `N: line`. */ + const readWindow = (lines: number): string => + `[src/app.ts#a1b2]\n${Array.from( + { length: lines }, + (_value, index) => `${index + 1}:${"x".repeat(200)}`, + ).join("\n")}`; + + const readCall = (id: string, path: string) => ({ + role: "assistant", + content: [{ type: "toolCall", id, name: "Read", arguments: { file_path: path } }], + }); + const readResult = (id: string, text: string) => ({ + role: "toolResult", + toolCallId: id, + toolName: "Read", + content: [{ type: "text", text }], + }); + const budget = (tokens: number) => ({ + tokens, + hardLimit: 224_000, + requestHeadroom: 32_000, + keepRecentTokens: 44_800, + }); + + /** + * An old row, then nine newer file-touching calls. The working set is the + * newest eight paths, so `src/old.ts` is outside it and its row is eligible + * while its siblings are not. + */ + const viewWithOldRead = () => + [ + readCall("call-old", "src/old.ts"), + readResult("call-old", readWindow(200)), + ...Array.from({ length: 9 }, (_value, index) => [ + readCall(`call-${index}`, `src/f${index}.ts`), + readResult(`call-${index}`, "x".repeat(50)), + ]).flat(), + ] as never[]; + + it("leaves the view untouched below the pressure gate", async () => { + const runtime = createRuntime(); + vi.spyOn(runtime as any, "contextBudget").mockReturnValue(budget(10_000)); + const view = viewWithOldRead(); + + expect((runtime as any).narrowToolResultsUnderPressure(view)).toBe(view); + await runtime.dispose(); + }); + + it("tiers an old Read result once the context is under pressure", async () => { + const runtime = createRuntime(); + vi.spyOn(runtime as any, "contextBudget").mockReturnValue(budget(160_000)); + const view = viewWithOldRead(); + const narrowed = (runtime as any).narrowToolResultsUnderPressure(view); + + expect(narrowed).not.toBe(view); + const first = (narrowed[1] as any).content[0].text as string; + expect(first).toContain("[tool result narrowed:"); + expect(first).toContain('Continue with Read path="src/old.ts" offset='); + // The rows the working set protects keep their identity. + expect(narrowed[3]).toBe(view[3]); + await runtime.dispose(); + }); + + it("tiers a spilled shell result without consulting the working set", async () => { + const runtime = createRuntime(); + vi.spyOn(runtime as any, "contextBudget").mockReturnValue(budget(160_000)); + const spill = + "y".repeat(30_000) + + "\n[truncated: kept the first 4000 of 51234 lines; limit 4000 lines / 96KB. " + + "Full output saved to C:\\scratch\\s1\\tool-output\\bash-1-1.log — " + + "Grep it, or Read it with offset/limit.]"; + const view = [ + { + role: "assistant", + content: [{ type: "toolCall", id: "call-1", name: "Bash", arguments: { command: "npm test" } }], + }, + { + role: "toolResult", + toolCallId: "call-1", + toolName: "Bash", + content: [{ type: "text", text: spill }], + }, + ...Array.from({ length: 6 }, () => ({ + role: "toolResult", + toolCallId: "recent", + toolName: "Bash", + content: [{ type: "text", text: "x".repeat(50) }], + })), + ] as never[]; + const narrowed = (runtime as any).narrowToolResultsUnderPressure(view); + + expect(narrowed).not.toBe(view); + const first = (narrowed[1] as any).content[0].text as string; + expect(first).toContain("tool-output\\bash-1-1.log"); + expect(first).toContain("Read it with offset/limit, or Grep it"); + await runtime.dispose(); + }); +}); diff --git a/packages/agent-runtime/src/runtime.ts b/packages/agent-runtime/src/runtime.ts index 1877654acd..40b455d84a 100644 --- a/packages/agent-runtime/src/runtime.ts +++ b/packages/agent-runtime/src/runtime.ts @@ -224,6 +224,11 @@ import { ContextEstimateCalibration, type ContextCalibration, } from "./context-calibration.js"; +import { + narrowToolResults, + TOOL_RESULT_TIER_PRESSURE, + workingSetPathsFrom, +} from "./tool-result-tier.js"; import { rebuildNodeNetworkTransport } from "./node-proxy.js"; import { @@ -1925,9 +1930,12 @@ Delegation rules: // The provider's rule that a tool-call id is unique is enforced here, on // the last view before the wire: the request is the only place it can be // guaranteed for both a rebuilt context and one that grew in this process. + // Old tool results are tiered in the same place: that pass is a no-op below + // its pressure gate and returns the same array when nothing qualified, so + // an ordinary request is byte-identical to what it was before it existed. convertToLlm: (messages) => alignRetainedReasoningIdentity( - convertToLlm(this.dropDuplicateToolCalls(messages)), + convertToLlm(this.narrowToolResultsUnderPressure(this.dropDuplicateToolCalls(messages))), this.reasoningReplayIdentity(), ), prepareNextTurnWithContext: (context, signal) => @@ -5828,11 +5836,39 @@ Delegation rules: additionalMessages: AgentMessage[] = [], ): boolean { const context = this.liveSessionContext(); - const messages = [...context.messages, ...additionalMessages]; + // Decide on the view the request would actually carry: narrowing old tool + // results shrinks it, and reading the un-narrowed projection here would + // make the saving invisible to this check, so compaction would fire while + // real room remained. + const messages = [ + ...this.narrowToolResultsUnderPressure(context.messages), + ...additionalMessages, + ]; const budget = this.contextBudget(messages); return this.compactionEnabled && budget.tokens >= budget.hardLimit; } + /** + * Shorten old tool results in an outgoing view, but only once the context is + * actually under pressure — below the gate the view comes back untouched, so + * the common case pays nothing. The full text of every narrowed result stays + * reachable through the pointer the pass embeds (see `tool-result-tier.ts`). + */ + private narrowToolResultsUnderPressure(messages: AgentMessage[]): AgentMessage[] { + if (messages.length === 0) return messages; + const budget = this.contextBudget(messages); + if (budget.tokens < budget.hardLimit * TOOL_RESULT_TIER_PRESSURE) { + return messages; + } + // The newest file-touching calls define what the task is about right now; a + // result for a file the session re-opened stays whole even when its row is + // old. `[keep]` immunity, the excluded tool families and the clear-at-least + // floor come from the module's own defaults. + return narrowToolResults(messages, { + workingSetPaths: workingSetPathsFrom(messages), + }).messages; + } + private retainedUserMessageBudget(budget: ContextBudget): number { return retainedUserMessageBudget(budget); } diff --git a/packages/agent-runtime/src/tool-result-tier.test.ts b/packages/agent-runtime/src/tool-result-tier.test.ts new file mode 100644 index 0000000000..01a26a7f9b --- /dev/null +++ b/packages/agent-runtime/src/tool-result-tier.test.ts @@ -0,0 +1,402 @@ +import type { AgentMessage } from "@earendil-works/pi-agent-core"; +import { describe, expect, it } from "vitest"; +import { + lastReadLine, + narrowToolResults, + spillPath, + TOOL_RESULT_TIER_CLEAR_AT_LEAST_CHARS, + TOOL_RESULT_TIER_EXCLUDE_TOOLS, + TOOL_RESULT_TIER_HEAD_CHARS, + TOOL_RESULT_TIER_KEEP_MARKERS, + TOOL_RESULT_TIER_KEEP_RECENT, + TOOL_RESULT_TIER_MIN_CHARS, + workingSetPathsFrom, +} from "./tool-result-tier.js"; + +const user = (text: string): AgentMessage => + ({ role: "user", content: text, timestamp: 1 }) as unknown as AgentMessage; + +const assistant = (text: string): AgentMessage => + ({ + role: "assistant", + content: [{ type: "text", text }], + provider: "p", + model: "m", + stopReason: "stop", + }) as unknown as AgentMessage; + +/** An assistant tool call in the shape the stored messages use. */ +const toolCall = ( + id: string, + name: string, + args: Record, + container: "arguments" | "args" = "arguments", +): AgentMessage => + ({ + role: "assistant", + content: [{ type: "toolCall", id, name, [container]: args }], + provider: "p", + model: "m", + stopReason: "toolUse", + }) as unknown as AgentMessage; + +const toolResult = ( + text: string, + extra: Record = {}, +): AgentMessage => + ({ + role: "toolResult", + toolCallId: "call-1", + toolName: "Read", + content: [{ type: "text", text }], + isError: false, + timestamp: 2, + ...extra, + }) as unknown as AgentMessage; + +/** A `Read` payload the way host-core renders one: header, then `N: line`. */ +function readWindow(lines: number, lineChars = 200): string { + const body = Array.from( + { length: lines }, + (_value, index) => `${index + 1}:${"x".repeat(lineChars)}`, + ).join("\n"); + return `[src/app.ts#a1b2]\n${body}`; +} + +/** A shell result whose output was spilled by the host's truncation marker. */ +function spilledShell(lines: number): string { + const body = Array.from({ length: lines }, () => "y".repeat(200)).join("\n"); + return ( + `${body}\n` + + "[truncated: kept the first 4000 of 51234 lines; limit 4000 lines / 96KB. " + + "Full output saved to C:\\data\\scratch\\s1\\tool-output\\bash-1-1.log — " + + "Grep it, or Read it with offset/limit.]" + ); +} + +const big = (chars: number): string => "x".repeat(chars); + +/** Old oversized result first, then the protected recent ones. */ +function oldReadWithRecentTail( + lines = 200, + extra: Record = {}, +): AgentMessage[] { + return [ + toolCall("call-1", "Read", { file_path: "src/app.ts" }), + toolResult(readWindow(lines), extra), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; +} + +const textAt = (list: AgentMessage[], index: number): string => { + const content = (list[index] as unknown as { content: unknown }).content; + if (typeof content === "string") return content; + return (content as Array<{ text: string }>)[0].text; +}; + +describe("recovery pointers", () => { + it("reads the last file line a kept head shows, ignoring digits inside a line", () => { + expect(lastReadLine("[p#a1b2]\n1:a\n2:b\n3:c")).toBe(3); + // A file line that itself begins with digits and a colon: only the + // rendered prefix counts, so the answer is still the row number. + expect(lastReadLine("[p#a1b2]\n41:9: inner\n42:x")).toBe(42); + // No numbered row at all: there is nothing to continue from. + expect(lastReadLine("[p#a1b2]\nbody")).toBeUndefined(); + }); + + it("finds the spill path in the host's own truncation sentence", () => { + expect( + spillPath( + "tail\n[truncated: kept the last 10 of 99 lines; limit 4000 lines / 96KB. " + + "Full output saved to /tmp/scratch/s1/tool-output/bash-1-1.log — Grep it, or Read it with offset/limit.]", + ), + ).toBe("/tmp/scratch/s1/tool-output/bash-1-1.log"); + // No spill, so no recovery path to name. + expect(spillPath("[truncated: kept the first 10 of 99 lines; limit 4000 lines / 96KB. Narrow the request to see more.]")).toBeUndefined(); + expect(spillPath("ordinary output")).toBeUndefined(); + }); +}); + +describe("narrowToolResults", () => { + it("shortens an old Read result and points at the next line of that file", () => { + const messages = oldReadWithRecentTail(); + const result = narrowToolResults(messages); + + expect(result.narrowed).toBe(1); + const narrowed = textAt(result.messages, 1); + expect(narrowed.length).toBeLessThan(TOOL_RESULT_TIER_HEAD_CHARS + 400); + expect(narrowed).toContain("[tool result narrowed:"); + expect(narrowed).toContain('Continue with Read path="src/app.ts" offset='); + // The pointer names a real continuation: the head ends at row N, so the + // reader resumes at N + 1. + const lastLine = lastReadLine(narrowed); + expect(lastLine).toBeDefined(); + expect(narrowed).toContain(`offset=${(lastLine ?? 0) + 1}`); + // Untouched rows come back by reference, so nothing else was rewritten. + expect(result.messages[0]).toBe(messages[0]); + expect(result.messages[2]).toBe(messages[2]); + }); + + it("shortens a spilled shell result and points at the spill file", () => { + const messages = [ + toolCall("call-1", "Bash", { command: "npm test" }), + toolResult(spilledShell(200), { toolName: "Bash" }), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50), { toolName: "Bash" }), + ), + ]; + const result = narrowToolResults(messages); + + expect(result.narrowed).toBe(1); + const narrowed = textAt(result.messages, 1); + expect(narrowed).toContain("[tool result narrowed:"); + expect(narrowed).toContain( + "C:\\data\\scratch\\s1\\tool-output\\bash-1-1.log", + ); + expect(narrowed).toContain("Read it with offset/limit, or Grep it"); + }); + + it("leaves a result whole when no recovery path can be named", () => { + const messages = [ + // A Grep result: re-running it is not guaranteed to return the same + // lines, so the pointer would not be exact. + toolCall("call-1", "Grep", { pattern: "needle" }), + toolResult(big(20_000), { toolName: "Grep" }), + // A shell result that was never spilled: nothing holds a fuller copy. + toolCall("call-2", "Bash", { command: "ls -R" }), + toolResult(big(20_000), { toolName: "Bash", toolCallId: "call-2" }), + // A Read result whose call named no path. + toolCall("call-3", "Read", {}), + toolResult(readWindow(200), { toolCallId: "call-3" }), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + const result = narrowToolResults(messages); + + expect(result.narrowed).toBe(0); + expect(result.savedChars).toBe(0); + expect(result.messages).toBe(messages); + }); + + it("leaves a result with more than one text block whole", () => { + const messages = [ + toolCall("call-1", "Read", { file_path: "src/app.ts" }), + { + role: "toolResult", + toolCallId: "call-1", + toolName: "Read", + content: [ + { type: "text", text: readWindow(100) }, + { type: "text", text: readWindow(100) }, + ], + timestamp: 2, + } as unknown as AgentMessage, + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + const result = narrowToolResults(messages); + + expect(result.narrowed).toBe(0); + expect(result.messages).toBe(messages); + }); + + it("never shortens user messages, assistant prose, or small results", () => { + const messages = [ + user(big(20_000)), + assistant(big(20_000)), + toolCall("call-1", "Read", { file_path: "src/app.ts" }), + toolResult(readWindow(3)), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + const result = narrowToolResults(messages); + + expect(result.narrowed).toBe(0); + expect(result.messages).toBe(messages); + }); + + it("keeps tool-search activation evidence intact", () => { + const messages = [ + toolCall("call-1", "Read", { file_path: "src/app.ts" }), + toolResult(readWindow(200), { addedToolNames: ["Read"] }), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + expect(narrowToolResults(messages).narrowed).toBe(0); + }); + + it("never shortens a result whose text carries a keep marker", () => { + expect(TOOL_RESULT_TIER_KEEP_MARKERS).toContain("[keep]"); + const body = `${readWindow(200)}\n[KEEP]`; + const messages = [ + toolCall("call-1", "Read", { file_path: "src/app.ts" }), + toolResult(body), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + expect(narrowToolResults(messages).narrowed).toBe(0); + }); + + it("never shortens results of excluded tools, blind to a missing tool name", () => { + expect(TOOL_RESULT_TIER_EXCLUDE_TOOLS).toContain("TaskWait"); + const excluded = [ + toolCall("call-1", "TaskWait", {}), + toolResult(readWindow(200), { toolName: "TaskWait" }), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + expect(narrowToolResults(excluded).narrowed).toBe(0); + + // Without a tool name the exclusion cannot match, but the pass then has no + // recovery path either, so the result still comes back whole. + const unnamed = [ + toolCall("call-1", "Read", { file_path: "src/app.ts" }), + toolResult(readWindow(200), { toolName: undefined }), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + expect(narrowToolResults(unnamed).narrowed).toBe(0); + }); + + it("keeps working-set results whole and narrows the rest", () => { + const messages = [ + toolCall("call-1", "Read", { file_path: "src/in-play.ts" }), + toolResult(readWindow(200), { toolCallId: "call-1" }), + toolCall("call-2", "Read", { file_path: "src/old.ts" }), + toolResult(readWindow(200), { toolCallId: "call-2" }), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + const result = narrowToolResults(messages, { + workingSetPaths: ["src/in-play.ts"], + }); + + expect(result.narrowed).toBe(1); + expect(textAt(result.messages, 1)).toBe(readWindow(200)); + expect(textAt(result.messages, 3)).toContain("[tool result narrowed:"); + }); + + it("works from the paths the batch's own calls named", () => { + const messages = [ + toolCall("a", "Read", { file_path: "one.ts" }, "args"), + toolCall("b", "Read", { path: "two.ts/" }), + toolCall("c", "Edit", { filePath: "C:\\work\\three.ts" }), + ]; + expect(workingSetPathsFrom(messages)).toEqual([ + "one.ts", + "two.ts", + "C:/work/three.ts", + ]); + }); + + it("returns the original array when the saving misses clearAtLeastChars", () => { + expect(TOOL_RESULT_TIER_CLEAR_AT_LEAST_CHARS).toBe(8_000); + const messages = [ + toolCall("call-1", "Read", { file_path: "src/app.ts" }), + // Over minChars, but narrowing it saves less than the floor. + toolResult(readWindow(25, 200)), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + const result = narrowToolResults(messages); + + expect(result.narrowed).toBe(0); + expect(result.savedChars).toBe(0); + expect(result.messages).toBe(messages); + }); + + it("reports the characters a pass saved, pointer included", () => { + const messages = oldReadWithRecentTail(); + const before = textAt(messages, 1).length; + const result = narrowToolResults(messages); + + expect(result.savedChars).toBeGreaterThanOrEqual( + TOOL_RESULT_TIER_CLEAR_AT_LEAST_CHARS, + ); + expect(result.savedChars).toBe(before - textAt(result.messages, 1).length); + }); + + it("is deterministic and idempotent", () => { + const first = narrowToolResults(oldReadWithRecentTail()); + const second = narrowToolResults(oldReadWithRecentTail()); + expect(textAt(second.messages, 1)).toBe(textAt(first.messages, 1)); + + // A second pass over an already-narrowed view is a no-op: the head is below + // minChars, so nothing else moves. + const again = narrowToolResults(first.messages); + expect(again.narrowed).toBe(0); + }); + + it("returns the same array when there are no tool results at all", () => { + const messages = [user("hi"), assistant("hello")]; + const result = narrowToolResults(messages); + expect(result.messages).toBe(messages); + expect(result.narrowed).toBe(0); + expect(result.savedChars).toBe(0); + }); + + it("appends the pointer once, at the end, when content is a string", () => { + const body = readWindow(200); + const messages = [ + toolCall("call-1", "Read", { file_path: "src/app.ts" }), + { + role: "toolResult", + toolCallId: "call-1", + toolName: "Read", + content: body, + timestamp: 2, + } as unknown as AgentMessage, + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + const result = narrowToolResults(messages); + const narrowed = textAt(result.messages, 1); + + expect(narrowed.endsWith("]")).toBe(true); + expect(narrowed.split("[tool result narrowed:").length - 1).toBe(1); + expect(narrowed.startsWith(body.slice(0, 100))).toBe(true); + }); + + it("uses the documented defaults when no option is passed", () => { + expect(TOOL_RESULT_TIER_MIN_CHARS).toBe(4_000); + expect(TOOL_RESULT_TIER_HEAD_CHARS).toBe(1_200); + expect(TOOL_RESULT_TIER_KEEP_RECENT).toBe(6); + + // A result just over the default minimum, with nothing keeping it: still a + // no-op below the clear-at-least floor. + const messages = [ + toolCall("call-1", "Read", { file_path: "src/app.ts" }), + toolResult(readWindow(25, 200)), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + expect(narrowToolResults(messages).narrowed).toBe(0); + + // The same shape, larger: it narrows, and only the oldest result moved. + const large = [ + toolCall("call-1", "Read", { file_path: "src/app.ts" }), + toolResult(readWindow(400, 200)), + ...Array.from({ length: TOOL_RESULT_TIER_KEEP_RECENT }, () => + toolResult(big(50)), + ), + ]; + const result = narrowToolResults(large); + expect(result.narrowed).toBe(1); + for (let index = 2; index < large.length; index += 1) { + expect(result.messages[index]).toBe(large[index]); + } + }); +}); diff --git a/packages/agent-runtime/src/tool-result-tier.ts b/packages/agent-runtime/src/tool-result-tier.ts new file mode 100644 index 0000000000..98af05b858 --- /dev/null +++ b/packages/agent-runtime/src/tool-result-tier.ts @@ -0,0 +1,508 @@ +/** + * Tool-result tiering for the outgoing view. + * + * Tool results are most of the context mass. Under pressure an old tool result + * is the cheapest thing to shorten: its text is not lost — a file-backed result + * can be read again, and a spilled shell result is still on disk — so shortening + * it removes low-density weight without creating a gap the reader cannot close. + * Paying for a lossy summary pass to recover the same room is the expensive + * alternative. + * + * This complements the per-result budgets in + * `docs/spec/03-runtime/16-tool-result-limits.md` rather than replacing them. + * Those bound **one** result when it is produced (Read 128 KB, shell 96 KB, plus + * spill files); this pass bounds the **aggregate** an outgoing request carries, + * which several individually-legal results can still exceed. A 128 KB Read + * window narrowed to a head plus a pointer is exactly the mass that matters + * here. + * + * The pass is deliberately narrow: + * + * - **Only tool results.** User messages, assistant prose and the compaction + * summary are never touched: they are synthesis or intent, and there is no + * way to fetch them again. + * - **Only the outgoing view.** Callers pass the message list a provider request + * is about to send; stored messages are never modified, and the projection is + * rebuilt for every request, so a later request with room renders the full + * text again. + * - **Deterministic.** Fixed size and count thresholds. No model call, no + * scoring, no per-session state. Every decision reads only the batch handed in. + * - **A recovery path or nothing.** A result is narrowed only when this module + * can name a real way to get the rest back, and the pointer names it exactly: + * a file-backed `Read` result points at the next line of that file, and a + * shell result whose output was spilled points at the spill file. Anything + * else — a `Grep`, a `Glob`, a shell result that was not spilled — is left + * whole, because a pointer the reader cannot act on is worse than a long + * result. That is also why `Grep`/`Glob` are not narrowed: re-running them is + * not guaranteed to return the same lines, so the pointer would not be exact. + * - **A no-op by default.** The caller decides when to run it (see + * `TOOL_RESULT_TIER_PRESSURE`); when nothing qualifies, or when the saving is + * below the floor, the same array comes back. + * + * Four precision knobs narrow the pass further. Each defaults to the documented + * value: + * + * - `keepMarkers` — a result whose text contains any marker is never shortened, + * so a user-pasted error or a result the model flagged to keep stays whole. + * Matched with a **case-insensitive plain substring**, never a regex: result + * text is user content and its metacharacters must not be interpreted. + * - `excludeTools` — results of these tools are never shortened. Subagent + * reports and retrieval hits are the most expensive information in the window; + * shortening them throws away what was just fetched. Matched by **exact, + * case-sensitive tool name**. + * - `workingSetPaths` — a result whose tool call named one of these files stays + * whole: a file the session is still working in is still in play. The + * `toolCallId → path` map is built from the `toolCall` blocks of the **same + * batch**, never from stored messages. Paths compare after normalizing + * trailing slashes, path separators and the Windows drive-letter case — no + * `realpath`, no filesystem access. + * - `clearAtLeastChars` — the pass is a no-op unless its summed saving reaches + * this floor. Rewriting the outgoing view for a few hundred characters is a + * bad trade: it breaks the provider's prompt cache for everything after the + * rewrite. + * +/** + * Fraction of the hard context limit at which the caller should start tiering. + * Below the compaction threshold on purpose: the point is to drop back under it + * with cheap, recoverable evidence instead of paying for a lossy summary pass. + */ +export const TOOL_RESULT_TIER_PRESSURE = 0.6; + +/** Results at or below this size are not worth a pointer. */ +import type { AgentMessage } from "@earendil-works/pi-agent-core"; + +/** Results at or below this size are not worth a pointer. */ +export const TOOL_RESULT_TIER_MIN_CHARS = 4_000; + +/** How much of a narrowed result stays inline. */ +export const TOOL_RESULT_TIER_HEAD_CHARS = 1_200; + +/** The newest results are almost certainly still in play; leave them whole. */ +export const TOOL_RESULT_TIER_KEEP_RECENT = 6; + +/** A result whose text contains any of these markers is never narrowed. */ +export const TOOL_RESULT_TIER_KEEP_MARKERS = ["[keep]"]; + +/** Results of these tools are never narrowed: they are the expensive parts. */ +export const TOOL_RESULT_TIER_EXCLUDE_TOOLS = [ + "Task", + "TaskWait", + "TaskList", + "TaskStop", + "ToolSearch", +]; + +/** + * The pass is a no-op unless it saves at least this many characters — a small + * rewrite is not worth breaking the provider's prompt cache. + */ +export const TOOL_RESULT_TIER_CLEAR_AT_LEAST_CHARS = 8_000; + +export type ToolResultTierOptions = { + minChars?: number; + headChars?: number; + keepRecent?: number; + /** Markers that immunize a result from narrowing (case-insensitive). */ + keepMarkers?: string[]; + /** Tool names whose results are never narrowed. */ + excludeTools?: string[]; + /** File paths whose tool results stay whole. Empty or absent means none. */ + workingSetPaths?: string[]; + /** Minimum summed saving for the pass to apply at all. */ + clearAtLeastChars?: number; +}; + +export type ToolResultTierResult = { + messages: AgentMessage[]; + /** How many results were shortened (0 means the same array came back). */ + narrowed: number; + /** Characters this pass removed, pointer text included; 0 when it no-opped. */ + savedChars: number; +}; + +type TextBlock = { type?: unknown; text?: unknown }; + +type ToolResultRecord = { + role: string; + content: unknown; + addedToolNames?: unknown; + toolCallId?: unknown; + toolName?: unknown; +}; + +/** Characters of text across a message's text blocks. */ +function textChars(content: unknown): number { + if (typeof content === "string") return content.length; + if (!Array.isArray(content)) return 0; + let total = 0; + for (const block of content as TextBlock[]) { + if (block && typeof block === "object" && typeof block.text === "string") { + total += block.text.length; + } + } + return total; +} + +/** The single text payload of a result, when it has exactly one. */ +function singleText(content: unknown): string | undefined { + if (typeof content === "string") return content; + if (!Array.isArray(content)) return undefined; + const texts = (content as TextBlock[]) + .filter((block) => block && typeof block.text === "string") + .map((block) => block.text as string); + return texts.length === 1 ? texts[0] : undefined; +} + +/** + * The last file line number a `Read` window shows. + * + * The payload is `[path#tag]` followed by `N: line` rows whose `N` is the + * file's own line number, so the last one is where a continuation starts. Rows + * are scanned from the end because a file line may itself begin with digits and + * a colon; only the rendered prefix counts. + */ +export function lastReadLine(head: string): number | undefined { + const lines = head.split("\n"); + for (let index = lines.length - 1; index >= 0; index -= 1) { + const match = /^\s*(\d+):/.exec(lines[index] ?? ""); + if (match) return Number(match[1]); + } + return undefined; +} + +/** The spill file a shell result's truncation marker named, when it did. */ +export function spillPath(content: string): string | undefined { + // The sentence is host-core's (`tools/mod.rs`): "Full output saved to + // — Grep it, or Read it with offset/limit." The dash and the trailing clause + // are matched loosely so a rewrite of the wording is not a silent failure. + const match = /Full output saved to ([^\n]+?)\s+(?:—|–|--)\s+Grep it/.exec( + content, + ); + const path = match?.[1]?.trim(); + return path && path.length > 0 ? path : undefined; +} + +/** A pointer to text that can still be fetched, or `undefined` when it cannot. */ +export type RecoveryPointer = { + /** The exact call that reads the rest. */ + pointer: string; + /** Set when the pointer is a line offset into `path` (a `Read` result). */ + path?: string; + offset?: number; +}; + +/** + * The recovery path for one result, or `undefined` to leave it whole. + * + * `path` is the file its tool call named, when the call named one. + */ +export function toolResultRecovery(input: { + toolName: string | undefined; + content: string; + head: string; + totalChars: number; + path: string | undefined; +}): RecoveryPointer | undefined { + const shown = input.head.length; + if (input.toolName === "Read" && input.path) { + const line = lastReadLine(input.head); + if (line === undefined) return undefined; + return { + path: input.path, + offset: line + 1, + pointer: + `[tool result narrowed: kept the first ${shown} of ${input.totalChars} characters, through line ${line}. ` + + `Continue with Read path="${input.path}" offset=${line + 1}.]`, + }; + } + const spill = spillPath(input.content); + if (spill) { + return { + pointer: + `[tool result narrowed: kept the first ${shown} of ${input.totalChars} characters. ` + + `The full captured output is at ${spill} — Read it with offset/limit, or Grep it.]`, + }; + } + return undefined; +} + +/** + * Narrow a result whose text is one string or one text block: keep its head and + * append the pointer once, at its end. A result whose text spans more than one + * block is not eligible: a continuation offset cannot be an honest pointer when + * the kept head is only part of the payload. + */ +function narrowContent( + content: unknown, + headChars: number, + pointer: string, +): { content: unknown; changed: boolean } { + if (typeof content === "string") { + if (content.length <= headChars) return { content, changed: false }; + return { + content: `${content.slice(0, headChars)}${pointer}`, + changed: true, + }; + } + if (!Array.isArray(content)) return { content, changed: false }; + const blocks = content as Array>; + const textIndexes = blocks + .map((block, index) => (block && typeof block.text === "string" ? index : -1)) + .filter((index) => index >= 0); + if (textIndexes.length !== 1) return { content, changed: false }; + const textIndex = textIndexes[0]; + let changed = false; + const next = blocks.map((block, index) => { + if (typeof block.text !== "string" || index !== textIndex) return block; + if (block.text.length <= headChars) return block; + changed = true; + return { ...block, text: `${block.text.slice(0, headChars)}${pointer}` }; + }); + return changed ? { content: next, changed: true } : { content, changed: false }; +} + +/** Argument spellings a tool call may name a file by, in precedence order. */ +const FILE_PATH_ARG_KEYS = ["file_path", "path", "filePath"] as const; + +/** + * Normalize a path for set comparison: forward slashes, no trailing slash, + * upper-case Windows drive letter. Deliberately no `realpath` — no filesystem + * access, and the same input always compares the same way. + */ +function normalizePath(path: string): string { + let normalized = path.trim().replace(/\\/g, "/"); + while (normalized.length > 1 && normalized.endsWith("/")) { + normalized = normalized.slice(0, -1); + } + return normalized.replace( + /^([A-Za-z]):\//, + (_match, drive: string) => `${drive.toUpperCase()}:/`, + ); +} + +/** The first file path a tool call's argument object names, if any. */ +function pathFromArgs(args: unknown): string | undefined { + if (!args || typeof args !== "object") return undefined; + const record = args as Record; + for (const key of FILE_PATH_ARG_KEYS) { + const value = record[key]; + if (typeof value === "string" && value.length > 0) return value; + } + return undefined; +} + +/** + * `toolCallId` → path, read from the `toolCall` blocks of this same batch. + * Nothing here is persisted; the map exists for this pass only. + */ +function toolCallPaths(messages: AgentMessage[]): Map { + const paths = new Map(); + for (const message of messages) { + const record = message as unknown as { role?: unknown; content?: unknown }; + if (record.role !== "assistant" || !Array.isArray(record.content)) continue; + for (const block of record.content as Array>) { + if (!block || typeof block !== "object" || block.type !== "toolCall") { + continue; + } + const id = + typeof block.id === "string" + ? block.id + : typeof block.toolCallId === "string" + ? block.toolCallId + : undefined; + if (!id) continue; + // `args` is the spelling callers pass; `arguments` is what stored + // assistant messages use. Both name the same object. + const path = pathFromArgs(block.args ?? block.arguments); + if (path !== undefined) paths.set(id, normalizePath(path)); + } + } + return paths; +} + +/** + * How many of the newest file-touching tool calls define the working set. + * + * Eight, not the newest twenty: the pass already keeps the newest + * `KEEP_RECENT` results whole, so a wide working set mostly re-protects the same + * rows while making the pressure pass a no-op on small windows. + */ +export const TOOL_RESULT_TIER_WORKING_SET_TOOL_CALLS = 8; + +/** + * The working set the caller passes as `workingSetPaths`: the paths named by the + * newest `lastToolCalls` file-touching tool calls, newest call's path last. + * + * This pass keeps the newest N *results* whole, but that is a count, not a + * notion of "what the task is about right now". A result for a file the session + * has just re-opened is still in play even when its row is old. + */ +export function workingSetPathsFrom( + messages: AgentMessage[], + options: { lastToolCalls?: number } = {}, +): string[] { + const limit = Math.max( + 0, + Math.floor(options.lastToolCalls ?? TOOL_RESULT_TIER_WORKING_SET_TOOL_CALLS), + ); + if (limit === 0) return []; + const seen = new Set(); + const ordered: string[] = []; + // Walk backwards: the newest calls decide the set, and a path already taken + // keeps its newest position (an older row must not move it). + for ( + let index = messages.length - 1; + index >= 0 && seen.size < limit; + index -= 1 + ) { + const record = messages[index] as unknown as { + role?: unknown; + content?: unknown; + }; + if (record.role !== "assistant" || !Array.isArray(record.content)) continue; + const content = record.content as Array>; + for ( + let block = content.length - 1; + block >= 0 && seen.size < limit; + block -= 1 + ) { + const candidate = content[block]; + if ( + !candidate || + typeof candidate !== "object" || + candidate.type !== "toolCall" + ) { + continue; + } + const path = pathFromArgs(candidate.args ?? candidate.arguments); + if (path === undefined) continue; + const normalized = normalizePath(path); + if (seen.has(normalized)) continue; + seen.add(normalized); + ordered.push(normalized); + } + } + return ordered.reverse(); +} + +/** + * Whether any text in the content contains one of the (already lower-cased) + * markers, as a plain substring. No regex: result text is user content. + */ +function containsMarker(content: unknown, markers: string[]): boolean { + if (markers.length === 0) return false; + const texts: string[] = []; + const single = singleText(content); + if (single !== undefined) { + texts.push(single); + } else if (Array.isArray(content)) { + for (const block of content as TextBlock[]) { + if (block && typeof block === "object" && typeof block.text === "string") { + texts.push(block.text); + } + } + } + return texts.some((text) => { + const haystack = text.toLowerCase(); + return markers.some((marker) => haystack.includes(marker)); + }); +} + +/** + * Shorten old tool results in an outgoing view, naming the recovery path for + * each one. A result whose recovery path this module cannot name is skipped, + * because a pointer the reader cannot act on is worse than a long result. + * + * Returns the same array when nothing changed. + */ +export function narrowToolResults( + messages: AgentMessage[], + options: ToolResultTierOptions = {}, +): ToolResultTierResult { + const minChars = options.minChars ?? TOOL_RESULT_TIER_MIN_CHARS; + const headChars = options.headChars ?? TOOL_RESULT_TIER_HEAD_CHARS; + const keepRecent = options.keepRecent ?? TOOL_RESULT_TIER_KEEP_RECENT; + const keepMarkers = (options.keepMarkers ?? TOOL_RESULT_TIER_KEEP_MARKERS) + .filter((marker) => marker.length > 0) + .map((marker) => marker.toLowerCase()); + const excludeTools = new Set( + options.excludeTools ?? TOOL_RESULT_TIER_EXCLUDE_TOOLS, + ); + const clearAtLeastChars = + options.clearAtLeastChars ?? TOOL_RESULT_TIER_CLEAR_AT_LEAST_CHARS; + // The working set and the paths it is matched against both come from this + // batch alone; an unknown result is outside the set, never guessed at. + const workingSet = + options.workingSetPaths && options.workingSetPaths.length > 0 + ? new Set(options.workingSetPaths.map(normalizePath)) + : undefined; + const toolPaths = toolCallPaths(messages); + + // Which tool results are "recent" is decided over the whole list first, so + // the set does not depend on how many were narrowed. + const toolIndexes: number[] = []; + for (let index = 0; index < messages.length; index += 1) { + if ((messages[index] as { role?: unknown }).role === "toolResult") { + toolIndexes.push(index); + } + } + if (toolIndexes.length === 0) return { messages, narrowed: 0, savedChars: 0 }; + const protectedIndexes = new Set(toolIndexes.slice(-keepRecent)); + + let narrowed = 0; + let savedChars = 0; + const next = messages.map((message, index) => { + if (!toolIndexes.includes(index) || protectedIndexes.has(index)) { + return message; + } + const record = message as unknown as ToolResultRecord; + // Tool-search activation evidence lives on the result; shortening it would + // change what a later request may restore. + if (Array.isArray(record.addedToolNames) && record.addedToolNames.length > 0) { + return message; + } + // Subagent reports and retrieval hits are the least replaceable evidence. + if (typeof record.toolName === "string" && excludeTools.has(record.toolName)) { + return message; + } + // A file the session is still working in is still in play. + const callPath = + typeof record.toolCallId === "string" + ? toolPaths.get(record.toolCallId) + : undefined; + if (workingSet && callPath !== undefined && workingSet.has(callPath)) { + return message; + } + const totalChars = textChars(record.content); + if (totalChars <= minChars) return message; + // A user-pasted error or an explicitly preserved result stays byte whole. + if (containsMarker(record.content, keepMarkers)) return message; + const text = singleText(record.content); + if (text === undefined) return message; + const recovery = toolResultRecovery({ + toolName: typeof record.toolName === "string" ? record.toolName : undefined, + content: text, + head: text.slice(0, headChars), + totalChars, + path: callPath, + }); + if (!recovery) return message; + const narrowedContent = narrowContent( + record.content, + headChars, + recovery.pointer, + ); + if (!narrowedContent.changed) return message; + narrowed += 1; + savedChars += totalChars - textChars(narrowedContent.content); + return { ...record, content: narrowedContent.content } as unknown as AgentMessage; + }); + + if (narrowed === 0) return { messages, narrowed: 0, savedChars: 0 }; + // A rewrite costs the provider's prompt cache for every message after it; + // below the floor the trade is not worth taking, so the view stays as it was. + if (savedChars < clearAtLeastChars) { + return { messages, narrowed: 0, savedChars: 0 }; + } + return { messages: next, narrowed, savedChars }; +}