Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 24 additions & 6 deletions apps/desktop/src/components/ContextUsageInspector.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@ export function ContextUsageInspector({
responseDurationMs,
responseOutputTokens,
responseOutputEstimated = false,
estimatedOccupancyTokens,
}: {
usage: MessageUsage;
turnUsage: MessageUsage;
Expand All @@ -48,6 +49,12 @@ export function ContextUsageInspector({
responseDurationMs?: number;
responseOutputTokens?: number;
responseOutputEstimated?: boolean;
/**
* Post-compaction occupancy from the newest checkpoint, while it is newer
* than the latest usage-bearing message. The ring leads with it; the
* popover's provider rows keep the last real request.
*/
estimatedOccupancyTokens?: number;
}) {
const { t } = useTranslation();
const panelId = useId();
Expand All @@ -63,7 +70,18 @@ export function ContextUsageInspector({
const [open, setOpen] = useState(false);
const [popoverPosition, setPopoverPosition] =
useState<ContextInspectorPlacement | null>(null);
const context = calculateContextUsage(usage, contextWindow);
// A freshly compacted window has no request usage of its own: lead with the
// checkpoint's estimate until the next provider response lands.
const context =
estimatedOccupancyTokens !== undefined
? calculateContextUsage(
{ totalTokens: estimatedOccupancyTokens } as MessageUsage,
contextWindow,
)
: calculateContextUsage(usage, contextWindow);
// The estimate is not a measurement: mark it, so the number reads as an
// estimate rather than pretending to be the last request's real usage.
const occupancyPrefix = estimatedOccupancyTokens !== undefined ? "≈" : "";
// The display preference flips the leading figure only; capacity colors
// still follow remaining space so the warning state keeps one meaning.
const usageDisplay = useAppStore((state) =>
Expand Down Expand Up @@ -98,7 +116,7 @@ export function ContextUsageInspector({
// tooltip contract while `percent`/`count` stay numeric.
const ariaArguments = {
percent: display.percent,
count: formatCompactTokenCount(display.tokens),
count: occupancyPrefix + formatCompactTokenCount(display.tokens),
state:
display.display === "used"
? t("chat.usageContextAriaUsed")
Expand Down Expand Up @@ -261,14 +279,14 @@ export function ContextUsageInspector({
<strong className="context-inspector-heading-value">
{display.display === "used"
? t("chat.usageContextSpent", {
count: formatCompactTokenCount(display.tokens),
count: occupancyPrefix + formatCompactTokenCount(display.tokens),
})
: t("chat.usageContextLeft", {
count: formatCompactTokenCount(display.tokens),
count: occupancyPrefix + formatCompactTokenCount(display.tokens),
})}
</strong>
<strong className="context-inspector-heading-percent">
{display.percent}%
{occupancyPrefix}{display.percent}%
</strong>
</div>
<div className="context-inspector-window">
Expand Down Expand Up @@ -400,7 +418,7 @@ export function ContextUsageInspector({
/>
</svg>
<span className="context-inspector-ring-value">
{display.percent}%
{occupancyPrefix}{display.percent}%
</span>
</TooltipButton>
{popover && typeof document !== "undefined"
Expand Down
27 changes: 27 additions & 0 deletions apps/desktop/src/lib/latest-turn-context.ts
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,12 @@ export type LatestTurnContextInspector = {
responseDurationMs?: number;
responseOutputTokens?: number;
responseOutputEstimated: boolean;
/**
* Post-compaction occupancy estimate: present only while the newest
* checkpoint is newer than the latest usage-bearing message, because a
* freshly compacted window has no request of its own yet.
*/
estimatedOccupancyTokens?: number;
};

/**
Expand Down Expand Up @@ -53,6 +59,24 @@ export function latestTurnContextInspector(
.reverse()
.find((message) => message.usage);

// A freshly installed checkpoint knows its post-compaction occupancy; the
// ring leads with that estimate until a newer request reports usage.
const latestMark = compactions.at(-1);
const boundaryIndex = latestMark
? parentMessages.findIndex(
(message) => message.id === latestMark.throughMessageId,
)
: -1;
const usageIndex = latestUsageMessage
? parentMessages.indexOf(latestUsageMessage)
: -1;
const estimatedOccupancyTokens =
latestMark?.tokensAfter !== undefined &&
boundaryIndex >= 0 &&
boundaryIndex >= usageIndex
? Math.max(0, Math.round(latestMark.tokensAfter))
: undefined;

return {
// Occupancy and provider cache/input/output use this last request.
// turnUsage remains the visual-turn sum for completed-turn speed.
Expand All @@ -75,5 +99,8 @@ export function latestTurnContextInspector(
responseOutputEstimated: latestTurn
? assistantTurnResponseOutputIsEstimated(latestTurn)
: false,
...(estimatedOccupancyTokens !== undefined
? { estimatedOccupancyTokens }
: {}),
};
}
89 changes: 89 additions & 0 deletions apps/desktop/test/latest-turn-context.test.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -216,3 +216,92 @@ test("latest turn inspector keeps last-request usage beside the turn sum", () =>
assert.equal(inspector?.turnUsage.cacheReadTokens, 51_500);
assert.equal(inspector?.turnUsage.inputTokens, 57_300);
});

test("latest turn inspector leads with the checkpoint estimate after compaction", () => {
const inspector = latestTurnContextInspector(
[
message("u1", "user", "overflow"),
message("a1", "assistant", "done", {
providerId: "provider",
modelId: "catalog-model",
usage: { inputTokens: 150_000, outputTokens: 2_000, totalTokens: 152_000 },
}),
],
providerModels,
providers,
[
{
id: "mark-1",
generation: 1,
summaryTokens: 40,
tokensAfter: 24_000,
throughMessageId: "a1",
summarized: true,
},
],
);

assert.equal(inspector?.estimatedOccupancyTokens, 24_000);
// The real usage stays available for the popover's provider rows.
assert.equal(inspector?.usage.totalTokens, 152_000);
});

test("latest turn inspector drops the estimate once newer usage arrives", () => {
const inspector = latestTurnContextInspector(
[
message("u1", "user", "overflow"),
message("a1", "assistant", "done", {
providerId: "provider",
modelId: "catalog-model",
usage: { inputTokens: 150_000, outputTokens: 2_000, totalTokens: 152_000 },
}),
message("u2", "user", "after compaction"),
message("a2", "assistant", "fresh", {
providerId: "provider",
modelId: "catalog-model",
usage: { inputTokens: 26_000, outputTokens: 900, totalTokens: 26_900 },
}),
],
providerModels,
providers,
[
{
id: "mark-1",
generation: 1,
summaryTokens: 40,
tokensAfter: 24_000,
throughMessageId: "a1",
summarized: true,
},
],
);

assert.equal(inspector?.estimatedOccupancyTokens, undefined);
assert.equal(inspector?.usage.totalTokens, 26_900);
});

test("latest turn inspector ignores marks without an estimate", () => {
const inspector = latestTurnContextInspector(
[
message("u1", "user", "overflow"),
message("a1", "assistant", "done", {
providerId: "provider",
modelId: "catalog-model",
usage: { inputTokens: 150_000, outputTokens: 2_000, totalTokens: 152_000 },
}),
],
providerModels,
providers,
[
{
id: "mark-1",
generation: 1,
summaryTokens: 40,
throughMessageId: "a1",
summarized: true,
},
],
);

assert.equal(inspector?.estimatedOccupancyTokens, undefined);
});
1 change: 1 addition & 0 deletions crates/host-core/src/sessions.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3702,6 +3702,7 @@ mod tests {
first_kept_message_id: Some(first.into()),
through_message_id: through.into(),
tokens_before: 120_000,
tokens_after: None,
usage: None,
retained_tail: Some(json!([{
"role": "user",
Expand Down
40 changes: 40 additions & 0 deletions crates/host-core/src/transcripts.rs
Original file line number Diff line number Diff line change
Expand Up @@ -65,6 +65,15 @@ pub struct CompactionRecord {
pub first_kept_message_id: Option<String>,
pub through_message_id: String,
pub tokens_before: i64,
/// Occupancy the checkpoint itself believes the next request will carry,
/// stamped when the checkpoint is installed (`persistCheckpoint`). The
/// visible transcript is untouched by a compaction, so nothing else on disk
/// records what the model context shrank to — without this the context ring
/// keeps showing the pre-compaction request until the next one lands.
/// Optional: a transcript written before this field existed loads with
/// `None`, and a line carrying it stays readable by older readers.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub tokens_after: Option<i64>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub usage: Option<Value>,
#[serde(default, skip_serializing_if = "Option::is_none")]
Expand Down Expand Up @@ -1255,6 +1264,7 @@ mod tests {
first_kept_message_id: Some("m1".into()),
through_message_id: "m2".into(),
tokens_before: 42_000,
tokens_after: Some(21_000),
usage: Some(json!({ "input": 100, "output": 20 })),
retained_tail: Some(json!([{ "role": "user", "content": "again", "timestamp": 1 }])),
details: None,
Expand Down Expand Up @@ -1607,6 +1617,36 @@ mod tests {
assert_eq!(restored[0].id, "compact-1");
assert_eq!(restored[0].through_message_id, "m2");
assert_eq!(restored[0].tokens_before, 42_000);
// The post-compaction estimate rides the record so the context ring can
// report the new window without waiting for the next provider request.
assert_eq!(restored[0].tokens_after, Some(21_000));
}

/// A checkpoint written before the post-compaction estimate existed must
/// keep loading. The field is optional on the wire, so an old line has no
/// `tokensAfter` and readers that predate it ignore the new one.
#[test]
fn a_checkpoint_without_a_post_compaction_estimate_still_loads() {
let dir = tempdir().unwrap();
let path = transcript_path(dir.path(), "s1").unwrap();
std::fs::create_dir_all(path.parent().unwrap()).unwrap();
let header = header_line("s1", "2026-07-26T00:00:00Z").unwrap();
let legacy = json!({
"type": "compaction",
"id": "legacy-1",
"summary": "summary",
"throughMessageId": "m1",
"tokensBefore": 42_000,
"createdAt": "2026-07-26T00:00:02Z"
})
.to_string();
std::fs::write(&path, format!("{header}\n{legacy}\n")).unwrap();

let restored = read_compactions(dir.path(), "s1").unwrap();
assert_eq!(restored.len(), 1);
assert_eq!(restored[0].id, "legacy-1");
assert_eq!(restored[0].tokens_before, 42_000);
assert_eq!(restored[0].tokens_after, None);
}

#[test]
Expand Down
3 changes: 3 additions & 0 deletions packages/agent-runtime/src/runtime.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5921,6 +5921,9 @@ describe("DesktopAgentRuntime inline context compaction", () => {
throughMessageId: "recent-user",
generation: 1,
summaryTokens: 7,
// Stamped by persistCheckpoint so the ring can lead with the new
// window before any request reports it.
tokensAfter: expect.any(Number),
summarized: true,
},
}),
Expand Down
13 changes: 13 additions & 0 deletions packages/agent-runtime/src/runtime.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6373,6 +6373,19 @@ Delegation rules:
) {
return "oversized";
}

// The estimate of the compacted projection is smaller than what the next
// request will actually carry: it is built from message content alone, while
// the request also pays for the system prompt and the tool schemas.
// `tokensBefore` is a measured request size, so the gap between it and the
// pre-compaction estimate is that overhead — adding it back keeps this
// number comparable with the request-based occupancy shown elsewhere.
const preCompactionTokens = this.contextBudget(
this.liveSessionContext().messages,
).tokens;
checkpoint.tokensAfter =
compactedBudget.tokens +
Math.max(0, checkpoint.tokensBefore - preCompactionTokens);
try {
await this.host.call("session.appendCompaction", {
sessionId: this.sessionId,
Expand Down
10 changes: 10 additions & 0 deletions packages/shared/src/context-compaction.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,16 @@ describe("contextCompactionMark", () => {
});
});

it("carries the post-compaction estimate when the record has one", () => {
// The ring leads with this number until a real request reports usage, so it
// has to survive the record → mark hop the event is built from.
expect(contextCompactionMark(record({ tokensAfter: 21_000 }))).toMatchObject({
tokensAfter: 21_000,
});
// A checkpoint written before the field existed must not invent one.
expect(contextCompactionMark(record())).not.toHaveProperty("tokensAfter");
});

it("marks a rollover checkpoint as carrying no real summary", () => {
expect(
contextCompactionMark(record({ details: { strategy: "fresh_window" } }))
Expand Down
5 changes: 4 additions & 1 deletion packages/shared/src/context-compaction.ts
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,10 @@ export function contextCompactionMark(
throughMessageId: record.throughMessageId,
generation: checkpointGeneration(record.details),
summaryTokens: estimateSummaryTokens(record.summary ?? ""),
summarized: checkpointSummarized(record.details),
...(record.tokensAfter !== undefined
? { tokensAfter: record.tokensAfter }
: {}),
...(fallback ? { fallback } : {}),
summarized: checkpointSummarized(record.details),
};
}
10 changes: 10 additions & 0 deletions packages/shared/src/types/sessions.ts
Original file line number Diff line number Diff line change
Expand Up @@ -86,6 +86,14 @@ export type ContextCompactionRecord = {
firstKeptMessageId?: string;
throughMessageId: string;
tokensBefore: number;
/**
* Occupancy the checkpoint itself believes the next request will carry,
* stamped when the checkpoint is installed. A compaction leaves the message
* transcript untouched, so without this the context ring keeps showing the
* pre-compaction request until the next provider response lands. Optional:
* a checkpoint written before the field existed has none.
*/
tokensAfter?: number;
usage?: unknown;
retainedTail?: unknown[];
details?: unknown;
Expand Down Expand Up @@ -122,6 +130,8 @@ export type ContextCompactionMark = ContextCompactionStatus & {
* notice as a summary.
*/
fallback?: ContextCompactionFallback;
/** Carried from the record so the ring can lead with it (see above). */
tokensAfter?: number;
};

export type ContextCompactionReason = "manual" | "threshold" | "overflow";
Expand Down
Loading