Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
73 changes: 73 additions & 0 deletions client/src/hooks/Chat/__tests__/useTokenUsage.spec.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -138,6 +138,50 @@ describe('useTokenUsage — post-snapshot output', () => {
expect(result.current.runwayTurns).toBe(2);
});

it('never reports less than the instructions and messages the snapshot breaks down', () => {
/** Remaining was measured as if only the 2 message tokens were sent, while the
* same snapshot publishes a 57-token system prompt: 7 used would leave the
* breakdown a negative message share. */
const inconsistent = {
...tailSnapshot,
remainingContextTokens: 199998,
completedOutputTokens: 5,
breakdown: { maxContextTokens: 200000, instructionTokens: 57, messageTokens: 2 },
} as ContextSnapshot;
const { result } = renderTokenUsage(undefined, { snapshot: inconsistent });

expect(result.current.usedTokens).toBe(57 + 2 + 5);
});

it('includes the summary when a remaining-based count undercuts the breakdown', () => {
const inconsistent = {
...tailSnapshot,
remainingContextTokens: 199998,
completedOutputTokens: 5,
breakdown: {
...tailSnapshot.breakdown,
instructionTokens: 57,
summaryTokens: 11,
messageTokens: 2,
},
} as ContextSnapshot;
const { result } = renderTokenUsage(undefined, { snapshot: inconsistent });

expect(result.current.usedTokens).toBe(57 + 11 + 2 + 5);
expect(result.current.usedTokens - 57 - 11).toBe(7);
});

it('keeps the remaining-based count when it exceeds the breakdown', () => {
const withSummary = {
...tailSnapshot,
breakdown: { ...tailSnapshot.breakdown, summaryTokens: 700 },
} as ContextSnapshot;
const { result } = renderTokenUsage(undefined, { snapshot: withSummary });

/** 195000 pre-invoke used covers content the 14700-token breakdown omits. */
expect(result.current.usedTokens).toBe(197000);
});

it('counts the finalized output in what a summarization could reclaim', () => {
const { result } = renderTokenUsage();

Expand All @@ -157,6 +201,20 @@ describe('useTokenUsage — post-snapshot output', () => {
breakdown: { maxContextTokens: 200000, instructionTokens: 4000, messageTokens },
}) as unknown as ContextSnapshot;

const summarizedLegacySnapshot = (): ContextSnapshot => ({
...legacySnapshot(2),
anchorMessageId: 'a2',
completedOutputTokens: 5,
breakdown: { ...legacySnapshot(2).breakdown, instructionTokens: 57, summaryTokens: 11 },
});

it('includes the summary in a snapshot saved without remaining headroom', () => {
const { result } = renderTokenUsage(undefined, { snapshot: summarizedLegacySnapshot() });

expect(result.current.usedTokens).toBe(57 + 11 + 2 + 5);
expect(result.current.usedTokens - 57 - 11).toBe(7);
});

it('projects a branch whose snapshots predate the remaining-token field', () => {
/** Both readings are instruction+messages sums — 13000 then 14000, so the
* call grew 1000. Reading the absent remaining as zero would score both
Expand Down Expand Up @@ -245,6 +303,21 @@ describe('useTokenUsage — post-snapshot output', () => {
expect(result.current.runwayTurns).toBe(194);
});

it('restores the summary in a legacy snapshot saved on the viewed branch', () => {
const saved = summarizedLegacySnapshot();
const { result } = renderTokenUsage(new Map(), {
snapshot: { ...tailSnapshot, anchorMessageId: 'unrelated-branch' },
messages: messages.map((message) =>
message.messageId === 'a2'
? ({ ...message, metadata: { contextUsage: saved } } as TMessage)
: message,
),
});

expect(result.current.snapshot?.anchorMessageId).toBe('a2');
expect(result.current.usedTokens).toBe(57 + 11 + 2 + 5);
});

it('excludes the retained latest tool result from compaction savings', () => {
const { result } = renderTokenUsage(undefined, {
messages: messages.map((message) =>
Expand Down
11 changes: 9 additions & 2 deletions client/src/hooks/Chat/useTokenUsage.ts
Original file line number Diff line number Diff line change
Expand Up @@ -313,10 +313,17 @@ export default function useTokenUsage({
effective.remainingContextTokens != null
? normalizeTokenCount(effective.remainingContextTokens)
: null;
const breakdownUsed =
instructionTokens +
normalizeTokenCount(breakdown.summaryTokens) +
normalizeTokenCount(breakdown.messageTokens);
/** A remaining count measured against a smaller instruction total than the
* snapshot publishes would put used below the instruction and summary
* shares the breakdown subtracts, hiding the Messages row. */
const baseUsed =
remainingContextTokens != null
? maxTokens - remainingContextTokens
: instructionTokens + normalizeTokenCount(breakdown.messageTokens);
? Math.max(maxTokens - remainingContextTokens, breakdownUsed)
: breakdownUsed;
/** The snapshot is pre-invoke: in-flight output rides on `liveTokens` (0
* unless streaming this branch), the last call's finalized output on
* `completedOutputTokens`, and retained tool results on
Expand Down
9 changes: 7 additions & 2 deletions client/src/utils/tokens.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1016,14 +1016,19 @@ describe('per-message usage index (branch + total)', () => {
'a1',
{
contextBudget: 1000,
breakdown: { maxContextTokens: 1000, instructionTokens: 100, messageTokens: 60 },
breakdown: {
maxContextTokens: 1000,
instructionTokens: 100,
summaryTokens: 25,
messageTokens: 60,
},
},
],
/** Nothing to read at all — skipped rather than counted as a full window. */
['a2', { contextBudget: 1000, breakdown: { maxContextTokens: 1000 } }],
]);

expect(collectAnchorSeries(CONVO, 'a2', anchors)).toEqual([{ used: 160, basis: 'breakdown' }]);
expect(collectAnchorSeries(CONVO, 'a2', anchors)).toEqual([{ used: 185, basis: 'breakdown' }]);
});

it('latestExchangeTokens sums the tail response and its user turn', () => {
Expand Down
10 changes: 6 additions & 4 deletions client/src/utils/tokens.ts
Original file line number Diff line number Diff line change
Expand Up @@ -547,7 +547,7 @@ export function prunedBranchTokens(
/**
* One persisted snapshot's used-context reading, with the basis it was measured
* on. `remaining` is the authoritative pre-invoke figure (`budget − remaining`);
* `breakdown` is the instruction+messages sum a snapshot saved before
* `breakdown` is the instructions+summary+messages sum a snapshot saved before
* `remainingContextTokens` existed still supports. The two measure different
* quantities (the backend's remaining covers content the breakdown does not),
* so a growth delta may only compare readings of the same basis.
Expand Down Expand Up @@ -615,16 +615,18 @@ export function collectAnchorSeries(
snapshot.contextBudget ?? snapshot.breakdown?.maxContextTokens,
);
const configuration = snapshotConfiguration(snapshot);
/** Same precedence as the render path's `baseUsed`: the backend's
* remaining headroom when it was saved, else the breakdown sum. */
/** Runway growth keeps the raw remaining basis, even if the gauge floors
* that reading to the breakdown; older snapshots use all breakdown parts. */
if (snapshot.remainingContextTokens != null && budget > 0) {
const remaining = normalizeTokenCount(snapshot.remainingContextTokens);
series.push({ used: Math.max(0, budget - remaining), basis: 'remaining', configuration });
} else {
const used =
normalizeTokenCount(
snapshot.effectiveInstructionTokens ?? snapshot.breakdown?.instructionTokens,
) + normalizeTokenCount(snapshot.breakdown?.messageTokens);
) +
normalizeTokenCount(snapshot.breakdown?.summaryTokens) +
normalizeTokenCount(snapshot.breakdown?.messageTokens);
if (used > 0) {
series.push({ used, basis: 'breakdown', configuration });
}
Expand Down
Loading