From ce620f07b7713f6f21a61f65ea4dbea405097629 Mon Sep 17 00:00:00 2001 From: zhangqingkun976 <1047045074@QQ.COM> Date: Tue, 22 Sep 2026 09:34:12 +0800 Subject: [PATCH] feat(desktop): show the context ring what a compaction just freed Rebased onto main (79cd7ae8); the rebase applied cleanly, no conflicts. A compaction leaves the message transcript untouched by design, so the ring kept showing the pre-compaction request until the next provider response landed. `CompactionRecord.tokensAfter` (optional; a checkpoint written before the field existed reads as absent) carries the post-compaction estimate, the mark and the store carry it, and the inspector leads with it until a newer real request reports usage. C:\Users\10470\.pi-desktop\scratch\cb9e2d41-8a55-4a18-899d-92ca2791ab51\commit-700-r2.txt --- .../src/components/ContextUsageInspector.tsx | 30 +++++-- apps/desktop/src/lib/latest-turn-context.ts | 27 ++++++ .../desktop/test/latest-turn-context.test.mjs | 89 +++++++++++++++++++ crates/host-core/src/sessions.rs | 1 + crates/host-core/src/transcripts.rs | 40 +++++++++ packages/agent-runtime/src/runtime.test.ts | 3 + packages/agent-runtime/src/runtime.ts | 13 +++ .../shared/src/context-compaction.test.ts | 10 +++ packages/shared/src/context-compaction.ts | 5 +- packages/shared/src/types/sessions.ts | 10 +++ 10 files changed, 221 insertions(+), 7 deletions(-) diff --git a/apps/desktop/src/components/ContextUsageInspector.tsx b/apps/desktop/src/components/ContextUsageInspector.tsx index d1e36bb236..e08659a473 100644 --- a/apps/desktop/src/components/ContextUsageInspector.tsx +++ b/apps/desktop/src/components/ContextUsageInspector.tsx @@ -40,6 +40,7 @@ export function ContextUsageInspector({ responseDurationMs, responseOutputTokens, responseOutputEstimated = false, + estimatedOccupancyTokens, }: { usage: MessageUsage; turnUsage: MessageUsage; @@ -48,6 +49,12 @@ export function ContextUsageInspector({ responseDurationMs?: number; responseOutputTokens?: number; responseOutputEstimated?: boolean; + /** + * Post-compaction occupancy from the newest checkpoint, while it is newer + * than the latest usage-bearing message. The ring leads with it; the + * popover's provider rows keep the last real request. + */ + estimatedOccupancyTokens?: number; }) { const { t } = useTranslation(); const panelId = useId(); @@ -63,7 +70,18 @@ export function ContextUsageInspector({ const [open, setOpen] = useState(false); const [popoverPosition, setPopoverPosition] = useState(null); - const context = calculateContextUsage(usage, contextWindow); + // A freshly compacted window has no request usage of its own: lead with the + // checkpoint's estimate until the next provider response lands. + const context = + estimatedOccupancyTokens !== undefined + ? calculateContextUsage( + { totalTokens: estimatedOccupancyTokens } as MessageUsage, + contextWindow, + ) + : calculateContextUsage(usage, contextWindow); + // The estimate is not a measurement: mark it, so the number reads as an + // estimate rather than pretending to be the last request's real usage. + const occupancyPrefix = estimatedOccupancyTokens !== undefined ? "≈" : ""; // The display preference flips the leading figure only; capacity colors // still follow remaining space so the warning state keeps one meaning. const usageDisplay = useAppStore((state) => @@ -98,7 +116,7 @@ export function ContextUsageInspector({ // tooltip contract while `percent`/`count` stay numeric. const ariaArguments = { percent: display.percent, - count: formatCompactTokenCount(display.tokens), + count: occupancyPrefix + formatCompactTokenCount(display.tokens), state: display.display === "used" ? t("chat.usageContextAriaUsed") @@ -261,14 +279,14 @@ export function ContextUsageInspector({ {display.display === "used" ? t("chat.usageContextSpent", { - count: formatCompactTokenCount(display.tokens), + count: occupancyPrefix + formatCompactTokenCount(display.tokens), }) : t("chat.usageContextLeft", { - count: formatCompactTokenCount(display.tokens), + count: occupancyPrefix + formatCompactTokenCount(display.tokens), })} - {display.percent}% + {occupancyPrefix}{display.percent}%
@@ -400,7 +418,7 @@ export function ContextUsageInspector({ /> - {display.percent}% + {occupancyPrefix}{display.percent}% {popover && typeof document !== "undefined" diff --git a/apps/desktop/src/lib/latest-turn-context.ts b/apps/desktop/src/lib/latest-turn-context.ts index add34979e7..2f2c83e51c 100644 --- a/apps/desktop/src/lib/latest-turn-context.ts +++ b/apps/desktop/src/lib/latest-turn-context.ts @@ -24,6 +24,12 @@ export type LatestTurnContextInspector = { responseDurationMs?: number; responseOutputTokens?: number; responseOutputEstimated: boolean; + /** + * Post-compaction occupancy estimate: present only while the newest + * checkpoint is newer than the latest usage-bearing message, because a + * freshly compacted window has no request of its own yet. + */ + estimatedOccupancyTokens?: number; }; /** @@ -53,6 +59,24 @@ export function latestTurnContextInspector( .reverse() .find((message) => message.usage); + // A freshly installed checkpoint knows its post-compaction occupancy; the + // ring leads with that estimate until a newer request reports usage. + const latestMark = compactions.at(-1); + const boundaryIndex = latestMark + ? parentMessages.findIndex( + (message) => message.id === latestMark.throughMessageId, + ) + : -1; + const usageIndex = latestUsageMessage + ? parentMessages.indexOf(latestUsageMessage) + : -1; + const estimatedOccupancyTokens = + latestMark?.tokensAfter !== undefined && + boundaryIndex >= 0 && + boundaryIndex >= usageIndex + ? Math.max(0, Math.round(latestMark.tokensAfter)) + : undefined; + return { // Occupancy and provider cache/input/output use this last request. // turnUsage remains the visual-turn sum for completed-turn speed. @@ -75,5 +99,8 @@ export function latestTurnContextInspector( responseOutputEstimated: latestTurn ? assistantTurnResponseOutputIsEstimated(latestTurn) : false, + ...(estimatedOccupancyTokens !== undefined + ? { estimatedOccupancyTokens } + : {}), }; } diff --git a/apps/desktop/test/latest-turn-context.test.mjs b/apps/desktop/test/latest-turn-context.test.mjs index 7c0ae21cb0..87d2f04571 100644 --- a/apps/desktop/test/latest-turn-context.test.mjs +++ b/apps/desktop/test/latest-turn-context.test.mjs @@ -216,3 +216,92 @@ test("latest turn inspector keeps last-request usage beside the turn sum", () => assert.equal(inspector?.turnUsage.cacheReadTokens, 51_500); assert.equal(inspector?.turnUsage.inputTokens, 57_300); }); + +test("latest turn inspector leads with the checkpoint estimate after compaction", () => { + const inspector = latestTurnContextInspector( + [ + message("u1", "user", "overflow"), + message("a1", "assistant", "done", { + providerId: "provider", + modelId: "catalog-model", + usage: { inputTokens: 150_000, outputTokens: 2_000, totalTokens: 152_000 }, + }), + ], + providerModels, + providers, + [ + { + id: "mark-1", + generation: 1, + summaryTokens: 40, + tokensAfter: 24_000, + throughMessageId: "a1", + summarized: true, + }, + ], + ); + + assert.equal(inspector?.estimatedOccupancyTokens, 24_000); + // The real usage stays available for the popover's provider rows. + assert.equal(inspector?.usage.totalTokens, 152_000); +}); + +test("latest turn inspector drops the estimate once newer usage arrives", () => { + const inspector = latestTurnContextInspector( + [ + message("u1", "user", "overflow"), + message("a1", "assistant", "done", { + providerId: "provider", + modelId: "catalog-model", + usage: { inputTokens: 150_000, outputTokens: 2_000, totalTokens: 152_000 }, + }), + message("u2", "user", "after compaction"), + message("a2", "assistant", "fresh", { + providerId: "provider", + modelId: "catalog-model", + usage: { inputTokens: 26_000, outputTokens: 900, totalTokens: 26_900 }, + }), + ], + providerModels, + providers, + [ + { + id: "mark-1", + generation: 1, + summaryTokens: 40, + tokensAfter: 24_000, + throughMessageId: "a1", + summarized: true, + }, + ], + ); + + assert.equal(inspector?.estimatedOccupancyTokens, undefined); + assert.equal(inspector?.usage.totalTokens, 26_900); +}); + +test("latest turn inspector ignores marks without an estimate", () => { + const inspector = latestTurnContextInspector( + [ + message("u1", "user", "overflow"), + message("a1", "assistant", "done", { + providerId: "provider", + modelId: "catalog-model", + usage: { inputTokens: 150_000, outputTokens: 2_000, totalTokens: 152_000 }, + }), + ], + providerModels, + providers, + [ + { + id: "mark-1", + generation: 1, + summaryTokens: 40, + throughMessageId: "a1", + summarized: true, + }, + ], + ); + + assert.equal(inspector?.estimatedOccupancyTokens, undefined); +}); diff --git a/crates/host-core/src/sessions.rs b/crates/host-core/src/sessions.rs index f8e794fb41..0731621db7 100644 --- a/crates/host-core/src/sessions.rs +++ b/crates/host-core/src/sessions.rs @@ -3702,6 +3702,7 @@ mod tests { first_kept_message_id: Some(first.into()), through_message_id: through.into(), tokens_before: 120_000, + tokens_after: None, usage: None, retained_tail: Some(json!([{ "role": "user", diff --git a/crates/host-core/src/transcripts.rs b/crates/host-core/src/transcripts.rs index d0f3bf2974..eec8e5d365 100644 --- a/crates/host-core/src/transcripts.rs +++ b/crates/host-core/src/transcripts.rs @@ -65,6 +65,15 @@ pub struct CompactionRecord { pub first_kept_message_id: Option, pub through_message_id: String, pub tokens_before: i64, + /// Occupancy the checkpoint itself believes the next request will carry, + /// stamped when the checkpoint is installed (`persistCheckpoint`). The + /// visible transcript is untouched by a compaction, so nothing else on disk + /// records what the model context shrank to — without this the context ring + /// keeps showing the pre-compaction request until the next one lands. + /// Optional: a transcript written before this field existed loads with + /// `None`, and a line carrying it stays readable by older readers. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tokens_after: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub usage: Option, #[serde(default, skip_serializing_if = "Option::is_none")] @@ -1255,6 +1264,7 @@ mod tests { first_kept_message_id: Some("m1".into()), through_message_id: "m2".into(), tokens_before: 42_000, + tokens_after: Some(21_000), usage: Some(json!({ "input": 100, "output": 20 })), retained_tail: Some(json!([{ "role": "user", "content": "again", "timestamp": 1 }])), details: None, @@ -1607,6 +1617,36 @@ mod tests { assert_eq!(restored[0].id, "compact-1"); assert_eq!(restored[0].through_message_id, "m2"); assert_eq!(restored[0].tokens_before, 42_000); + // The post-compaction estimate rides the record so the context ring can + // report the new window without waiting for the next provider request. + assert_eq!(restored[0].tokens_after, Some(21_000)); + } + + /// A checkpoint written before the post-compaction estimate existed must + /// keep loading. The field is optional on the wire, so an old line has no + /// `tokensAfter` and readers that predate it ignore the new one. + #[test] + fn a_checkpoint_without_a_post_compaction_estimate_still_loads() { + let dir = tempdir().unwrap(); + let path = transcript_path(dir.path(), "s1").unwrap(); + std::fs::create_dir_all(path.parent().unwrap()).unwrap(); + let header = header_line("s1", "2026-07-26T00:00:00Z").unwrap(); + let legacy = json!({ + "type": "compaction", + "id": "legacy-1", + "summary": "summary", + "throughMessageId": "m1", + "tokensBefore": 42_000, + "createdAt": "2026-07-26T00:00:02Z" + }) + .to_string(); + std::fs::write(&path, format!("{header}\n{legacy}\n")).unwrap(); + + let restored = read_compactions(dir.path(), "s1").unwrap(); + assert_eq!(restored.len(), 1); + assert_eq!(restored[0].id, "legacy-1"); + assert_eq!(restored[0].tokens_before, 42_000); + assert_eq!(restored[0].tokens_after, None); } #[test] diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index cc0346e2bb..ceaef0e9f1 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -5921,6 +5921,9 @@ describe("DesktopAgentRuntime inline context compaction", () => { throughMessageId: "recent-user", generation: 1, summaryTokens: 7, + // Stamped by persistCheckpoint so the ring can lead with the new + // window before any request reports it. + tokensAfter: expect.any(Number), summarized: true, }, }), diff --git a/packages/agent-runtime/src/runtime.ts b/packages/agent-runtime/src/runtime.ts index 1877654acd..8741695d86 100644 --- a/packages/agent-runtime/src/runtime.ts +++ b/packages/agent-runtime/src/runtime.ts @@ -6373,6 +6373,19 @@ Delegation rules: ) { return "oversized"; } + + // The estimate of the compacted projection is smaller than what the next + // request will actually carry: it is built from message content alone, while + // the request also pays for the system prompt and the tool schemas. + // `tokensBefore` is a measured request size, so the gap between it and the + // pre-compaction estimate is that overhead — adding it back keeps this + // number comparable with the request-based occupancy shown elsewhere. + const preCompactionTokens = this.contextBudget( + this.liveSessionContext().messages, + ).tokens; + checkpoint.tokensAfter = + compactedBudget.tokens + + Math.max(0, checkpoint.tokensBefore - preCompactionTokens); try { await this.host.call("session.appendCompaction", { sessionId: this.sessionId, diff --git a/packages/shared/src/context-compaction.test.ts b/packages/shared/src/context-compaction.test.ts index 7935083fa0..8082d12abe 100644 --- a/packages/shared/src/context-compaction.test.ts +++ b/packages/shared/src/context-compaction.test.ts @@ -56,6 +56,16 @@ describe("contextCompactionMark", () => { }); }); + it("carries the post-compaction estimate when the record has one", () => { + // The ring leads with this number until a real request reports usage, so it + // has to survive the record → mark hop the event is built from. + expect(contextCompactionMark(record({ tokensAfter: 21_000 }))).toMatchObject({ + tokensAfter: 21_000, + }); + // A checkpoint written before the field existed must not invent one. + expect(contextCompactionMark(record())).not.toHaveProperty("tokensAfter"); + }); + it("marks a rollover checkpoint as carrying no real summary", () => { expect( contextCompactionMark(record({ details: { strategy: "fresh_window" } })) diff --git a/packages/shared/src/context-compaction.ts b/packages/shared/src/context-compaction.ts index 55c5535666..e10b201ebc 100644 --- a/packages/shared/src/context-compaction.ts +++ b/packages/shared/src/context-compaction.ts @@ -53,7 +53,10 @@ export function contextCompactionMark( throughMessageId: record.throughMessageId, generation: checkpointGeneration(record.details), summaryTokens: estimateSummaryTokens(record.summary ?? ""), - summarized: checkpointSummarized(record.details), + ...(record.tokensAfter !== undefined + ? { tokensAfter: record.tokensAfter } + : {}), ...(fallback ? { fallback } : {}), + summarized: checkpointSummarized(record.details), }; } diff --git a/packages/shared/src/types/sessions.ts b/packages/shared/src/types/sessions.ts index 1196253962..e48aa3fe14 100644 --- a/packages/shared/src/types/sessions.ts +++ b/packages/shared/src/types/sessions.ts @@ -86,6 +86,14 @@ export type ContextCompactionRecord = { firstKeptMessageId?: string; throughMessageId: string; tokensBefore: number; + /** + * Occupancy the checkpoint itself believes the next request will carry, + * stamped when the checkpoint is installed. A compaction leaves the message + * transcript untouched, so without this the context ring keeps showing the + * pre-compaction request until the next provider response lands. Optional: + * a checkpoint written before the field existed has none. + */ + tokensAfter?: number; usage?: unknown; retainedTail?: unknown[]; details?: unknown; @@ -122,6 +130,8 @@ export type ContextCompactionMark = ContextCompactionStatus & { * notice as a summary. */ fallback?: ContextCompactionFallback; + /** Carried from the record so the ring can lead with it (see above). */ + tokensAfter?: number; }; export type ContextCompactionReason = "manual" | "threshold" | "overflow";