diff --git a/CHANGELOG.md b/CHANGELOG.md index 298d272dbe..40ef915ebe 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -40,6 +40,15 @@ ### Fixed +- The composer's context gauge now falls back to its localized `Usage` label + instead of holding the pre-compaction figure after a compaction: no provider + has tokenized the replacement prompt yet, so the last real count is stale. + A `context_compaction_applied` row marks the boundary the moment a + compaction lands — the `context_compacted` note stays the settlement-time + display row — and only a measurement that completed after the boundary + restores the figure, so a mid-turn compaction recovers as soon as the next + step settles. A failed-open compaction is not a boundary: that request went + out with its full raw history. - Fixed a renderer crash dialog reporting React error #185 ("Maximum update depth exceeded") coming from the composer's prompt-history inline completion (#4117): the offer engine the 0.1.11 composer fed could flip-flop its announcement state on diff --git a/apps/desktop/src/main/__tests__/conversation-owner.test.ts b/apps/desktop/src/main/__tests__/conversation-owner.test.ts index 69892c3280..202a9c9175 100644 --- a/apps/desktop/src/main/__tests__/conversation-owner.test.ts +++ b/apps/desktop/src/main/__tests__/conversation-owner.test.ts @@ -27,6 +27,7 @@ import type { DesktopSessionSummary } from '../../shared/desktop-session-project import { createSessionCatalogController, SessionCatalogContext } from '../../renderer/application/contracts/session-catalog/session-catalog-state.js'; import { ConversationProvider, ConversationServicesProvider, ConversationLifecycle, ConversationTranscriptRegion, ConversationComposerRegion, useAppShellSessionUiState, type ConversationObservationServices, type ConversationServices } from '../../renderer/features/conversation/index.js'; import { renderConversationMarkdown, stubComposerGateInputs, stubConversationServices, useConversationOwner } from '../../renderer/features/conversation/testing.js'; +import type { LatestRequestUsage } from '../../renderer/application/contracts/session-inspector/latest-request-usage.js'; import { cleanupFakeDom, installReactRenderer } from './fake-dom.js'; import { withComposerSubmission } from './composer-submission-fixture.js'; @@ -96,7 +97,7 @@ function harness(options: { let lifecycleCommits = 0; let setVisible!: (visible: boolean) => void; function Transcript(props: NonNullable) { transcript = props; transcriptRenders += 1; return null; } - function Composer(_props: { processing: boolean; pendingMessages?: readonly TransientUserMessageProjection[]; latestRequestUsageTokens?: number }) { + function Composer(_props: { processing: boolean; pendingMessages?: readonly TransientUserMessageProjection[]; latestRequestUsage?: LatestRequestUsage }) { composerRenders += 1; useEffect(() => { composerMounts += 1; return () => { composerUnmounts += 1; }; }, []); return null; diff --git a/apps/desktop/src/main/__tests__/latest-request-usage.test.ts b/apps/desktop/src/main/__tests__/latest-request-usage.test.ts index 301f8aeffa..069e582582 100644 --- a/apps/desktop/src/main/__tests__/latest-request-usage.test.ts +++ b/apps/desktop/src/main/__tests__/latest-request-usage.test.ts @@ -19,22 +19,55 @@ import assert from 'node:assert/strict'; import { test } from 'node:test'; -import { selectLatestRequestUsage } from '../../renderer/chat-composer-region.js'; +import { + resolveContextUsage, + selectLatestRequestUsage, +} from '../../renderer/application/contracts/session-inspector/latest-request-usage.js'; const ROUTE = { llmConnectionId: 'conn-a' }; const MODEL = 'model-a'; -function usage(anchor?: { - inputTokens: number; - outputTokens?: number; - modelId?: string; - connectionId?: string; -}) { - return { type: 'token_usage', ...(anchor ? { lastRequestAnchor: anchor } : {}) }; +function usage( + anchor?: { + inputTokens: number; + outputTokens?: number; + modelId?: string; + connectionId?: string; + completedAt?: number; + }, + ts?: number, +) { + return { + type: 'token_usage', + ...(ts !== undefined ? { ts } : {}), + ...(anchor ? { lastRequestAnchor: anchor } : {}), + }; +} + +function compactionNote(kind: string, ts?: number) { + return { type: 'system_note', kind, ...(ts !== undefined ? { ts } : {}) }; +} + +function appliedNote(ts?: number, turnId?: string) { + return { + type: 'system_note', + kind: 'context_compaction_applied', + ...(ts !== undefined ? { ts } : {}), + ...(turnId !== undefined ? { turnId } : {}), + }; +} + +function displayNote(ts?: number, turnId?: string) { + return { + type: 'system_note', + kind: 'context_compacted', + ...(ts !== undefined ? { ts } : {}), + ...(turnId !== undefined ? { turnId } : {}), + }; } test('reads the newest anchor on the active route', () => { - const tokens = selectLatestRequestUsage( + const reading = selectLatestRequestUsage( [ usage({ inputTokens: 10, outputTokens: 2, modelId: MODEL, connectionId: 'conn-a' }), { type: 'assistant' }, @@ -43,14 +76,14 @@ test('reads the newest anchor on the active route', () => { MODEL, ROUTE, ); - assert.equal(tokens, 120); + assert.deepEqual(reading, { kind: 'tokens', tokens: 120 }); }); test('scans past an anchorless usage row, which is what manual compaction writes', () => { // `/compact` appends a synthetic `token_usage` with no anchor. The runtime's // own reader skips it and keeps the last real request; stopping there would - // blank the indicator after every manual compaction. - const tokens = selectLatestRequestUsage( + // read the compaction's own record as a count of zero. + const reading = selectLatestRequestUsage( [ usage({ inputTokens: 100, outputTokens: 20, modelId: MODEL, connectionId: 'conn-a' }), usage(), @@ -58,37 +91,340 @@ test('scans past an anchorless usage row, which is what manual compaction writes MODEL, ROUTE, ); - assert.equal(tokens, 120); + assert.deepEqual(reading, { kind: 'tokens', tokens: 120 }); +}); + +test('a compaction boundary newer than every measurement supersedes it', () => { + // The compaction replaced the prompt the newest count described, and nothing has + // measured the replacement. The stale figure must not be shown as a live + // reading of what the session is about to send. + const reading = selectLatestRequestUsage( + [ + usage({ inputTokens: 100, outputTokens: 20, modelId: MODEL, connectionId: 'conn-a' }, 1_000), + compactionNote('context_compacted', 2_000), + usage(undefined, 2_100), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'compacted', at: 2_000 }); +}); + +test('a measurement newer than the boundary stands, which is the post-compaction reading', () => { + const reading = selectLatestRequestUsage( + [ + usage({ inputTokens: 100, outputTokens: 20, modelId: MODEL, connectionId: 'conn-a' }, 1_000), + compactionNote('context_compacted', 2_000), + usage({ inputTokens: 30, outputTokens: 5, modelId: MODEL, connectionId: 'conn-a', completedAt: 3_000 }, 3_100), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'tokens', tokens: 35, at: 3_000 }); +}); + +test('a post-compaction anchor supersedes a pre-compaction snapshot while diagnostics are pending', () => { + const latestRequestUsage = selectLatestRequestUsage( + [ + usage({ inputTokens: 90_000, modelId: MODEL, connectionId: 'conn-a' }, 1_000), + compactionNote('context_compacted', 2_000), + usage({ inputTokens: 35_000, modelId: MODEL, connectionId: 'conn-a', completedAt: 2_500 }, 3_000), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(latestRequestUsage, { kind: 'tokens', tokens: 35_000, at: 2_500 }); + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage, + live: { usageTokens: 90_000, contextWindow: 100_000, completedAt: 1_000 }, + }), + { kind: 'measured', tokens: 35_000 }, + ); +}); + +test('the token row written after its own request keeps the settled snapshot and window', () => { + const latestRequestUsage = selectLatestRequestUsage( + [usage({ inputTokens: 100, outputTokens: 20, modelId: MODEL, connectionId: 'conn-a', completedAt: 1_000 }, 1_100)], + MODEL, + ROUTE, + ); + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage, + live: { usageTokens: 100, contextWindow: 1_000, completedAt: 1_000 }, + }), + { kind: 'measured', tokens: 100, meteredWindow: 1_000 }, + ); +}); + +test('a legacy anchor without settlement time cannot displace a live snapshot', () => { + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage: selectLatestRequestUsage( + [{ + type: 'token_usage', ts: 1_100, + lastRequestAnchor: { inputTokens: 100, outputTokens: 20, modelId: MODEL, connectionId: 'conn-a' }, + }], + MODEL, + ROUTE, + ), + live: { usageTokens: 100, contextWindow: 1_000, completedAt: 1_000 }, + }), + { kind: 'measured', tokens: 100, meteredWindow: 1_000 }, + ); +}); + +test('a timed anchor wins when the retained snapshot has no settlement time', () => { + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage: { kind: 'tokens', tokens: 35_000, at: 3_000 }, + live: { usageTokens: 90_000 }, + }), + { kind: 'measured', tokens: 35_000 }, + ); +}); + +test('a failed-open compaction is not a boundary', () => { + // The compaction was refused and the request went out with its full raw history, so + // the measurement behind the note still describes what was sent. + const reading = selectLatestRequestUsage( + [ + usage({ inputTokens: 100, outputTokens: 20, modelId: MODEL, connectionId: 'conn-a' }), + compactionNote('context_compaction_failed_open'), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'tokens', tokens: 120 }); +}); + +test('a boundary with no measurement behind it is still a superseded reading', () => { + const reading = selectLatestRequestUsage([compactionNote('context_compacted')], MODEL, ROUTE); + assert.deepEqual(reading, { kind: 'compacted' }); +}); + +test('an apply-time boundary row supersedes with its own write time', () => { + // Mid-turn compactions record `context_compaction_applied` when the compaction lands, + // so the boundary time is the compaction moment even mid-turn. + const reading = selectLatestRequestUsage( + [ + usage({ inputTokens: 100, outputTokens: 20, modelId: MODEL, connectionId: 'conn-a' }, 1_000), + appliedNote(1_500, 'turn-1'), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'compacted', at: 1_500 }); +}); + +test('a settlement display note adopts the apply time of the compaction it describes', () => { + // The mid-turn case: `context_compaction_applied` is written when the compaction + // lands and `context_compacted` at settlement. The boundary time is the + // compaction's, so post-compaction measurements settling before the turn ends are not + // misjudged as pre-compaction. + const reading = selectLatestRequestUsage( + [ + usage({ inputTokens: 100, outputTokens: 20, modelId: MODEL, connectionId: 'conn-a' }, 1_000), + appliedNote(1_500, 'turn-1'), + displayNote(9_000, 'turn-1'), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'compacted', at: 1_500 }); +}); + +test('the latest apply row wins when compaction is applied twice in one turn', () => { + const reading = selectLatestRequestUsage( + [ + appliedNote(1_500, 'turn-1'), + appliedNote(4_000, 'turn-1'), + displayNote(9_000, 'turn-1'), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'compacted', at: 4_000 }); +}); + +test('an apply row from another turn does not lend the note its time', () => { + const reading = selectLatestRequestUsage( + [ + appliedNote(1_500, 'turn-1'), + displayNote(9_000, 'turn-2'), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'compacted', at: 9_000 }); +}); + +test('a usage row between the apply and the note keeps the note time', () => { + // Cannot arise in written data — the turn's usage row lands after the + // note — but the hunt stops at it rather than reaching across turns. + const reading = selectLatestRequestUsage( + [ + appliedNote(1_500, 'turn-1'), + usage({ inputTokens: 30, modelId: MODEL, connectionId: 'conn-a', completedAt: 2_000 }, 2_100), + displayNote(9_000, 'turn-1'), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'compacted', at: 9_000 }); +}); + +test('a usage row settled after the notes but anchored before the compaction is superseded (#5547)', () => { + // The failed-send settlement order: the apply row, then the display note, + // then the usage row — whose anchor is still the last COMPLETED request, + // which finished before the compaction because its retry never did. Position + // alone would misread the anchor as a post-compaction measurement. + const reading = selectLatestRequestUsage( + [ + appliedNote(1_500, 'turn-1'), + displayNote(9_000, 'turn-1'), + usage( + { + inputTokens: 100, + outputTokens: 20, + modelId: MODEL, + connectionId: 'conn-a', + completedAt: 1_000, + }, + 9_100, + ), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'compacted', at: 1_500 }); +}); + +test('a post-compaction measurement settled after the notes still stands', () => { + // Same ledger shape, healthy send: the retry completed after the compaction, so + // the anchor is genuinely post-compaction and stays the newest reading. + const reading = selectLatestRequestUsage( + [ + appliedNote(1_500, 'turn-1'), + displayNote(9_000, 'turn-1'), + usage( + { inputTokens: 30, modelId: MODEL, connectionId: 'conn-a', completedAt: 2_000 }, + 9_100, + ), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'tokens', tokens: 30, at: 2_000 }); +}); + +test('a boundary tied with the anchored completion cannot prove the anchor is post-compaction', () => { + const reading = selectLatestRequestUsage( + [ + appliedNote(1_500, 'turn-1'), + usage( + { inputTokens: 30, modelId: MODEL, connectionId: 'conn-a', completedAt: 1_500 }, + 9_100, + ), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'compacted', at: 1_500 }); +}); + +test('an anchor without a completion time cannot be superseded by a boundary behind it', () => { + const reading = selectLatestRequestUsage( + [ + appliedNote(1_500, 'turn-1'), + usage({ inputTokens: 30, modelId: MODEL, connectionId: 'conn-a' }, 9_100), + ], + MODEL, + ROUTE, + ); + assert.deepEqual(reading, { kind: 'tokens', tokens: 30 }); +}); + +test('the failed-retry ledger reads stale, not the pre-compaction measurement (#5547)', () => { + // End to end: the send compacted at 1_500 and its retry never completed, so the + // live snapshot and the settled anchor both describe the request that + // finished at 1_000 — pre-compaction context. The gauge must report stale. + const latestRequestUsage = selectLatestRequestUsage( + [ + appliedNote(1_500, 'turn-1'), + displayNote(9_000, 'turn-1'), + usage( + { + inputTokens: 190_000, + modelId: MODEL, + connectionId: 'conn-a', + completedAt: 1_000, + }, + 9_100, + ), + ], + MODEL, + ROUTE, + ); + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage, + live: { usageTokens: 190_000, contextWindow: 200_000, completedAt: 1_000 }, + }), + { kind: 'stale', reason: 'compaction' }, + ); +}); + +test('a post-compaction snapshot is measured against the apply time, not settlement', () => { + // The bug this event fixes: the compaction landed at 1_500, a later request + // settled at 2_000, and the turn itself settled at 9_000. Reading the + // display note's own write time would hide the valid post-compaction snapshot. + const latestRequestUsage = selectLatestRequestUsage( + [ + appliedNote(1_500, 'turn-1'), + displayNote(9_000, 'turn-1'), + ], + MODEL, + ROUTE, + ); + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage, + live: { usageTokens: 30_000, contextWindow: 100_000, completedAt: 2_000 }, + }), + { kind: 'measured', tokens: 30_000, meteredWindow: 100_000 }, + ); }); test('refuses an anchor from another model', () => { // A token count is a number in one model's tokenizer. Pairing model A's // count with model B's window produces a precise-looking figure about a // request the user is not making. - const tokens = selectLatestRequestUsage( + const reading = selectLatestRequestUsage( [usage({ inputTokens: 100_000, modelId: 'model-b', connectionId: 'conn-a' })], MODEL, ROUTE, ); - assert.equal(tokens, undefined); + assert.equal(reading, undefined); }); test('refuses an anchor from another connection', () => { - const tokens = selectLatestRequestUsage( + const reading = selectLatestRequestUsage( [usage({ inputTokens: 100, modelId: MODEL, connectionId: 'conn-b' })], MODEL, ROUTE, ); - assert.equal(tokens, undefined); + assert.equal(reading, undefined); }); test('refuses an anchor written before anchors carried their route', () => { - const tokens = selectLatestRequestUsage( + const reading = selectLatestRequestUsage( [usage({ inputTokens: 100, outputTokens: 20 })], MODEL, ROUTE, ); - assert.equal(tokens, undefined); + assert.equal(reading, undefined); }); test('refuses when there is no active route yet', () => { @@ -98,10 +434,108 @@ test('refuses when there is no active route yet', () => { }); test('refuses a non-positive input count', () => { - const tokens = selectLatestRequestUsage( + const reading = selectLatestRequestUsage( [usage({ inputTokens: 0, modelId: MODEL, connectionId: 'conn-a' })], MODEL, ROUTE, ); - assert.equal(tokens, undefined); + assert.equal(reading, undefined); +}); + +test('the snapshot is the reading when it is the newer answer', () => { + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage: { kind: 'tokens', tokens: 120, at: 1_000 }, + live: { usageTokens: 130, completedAt: 1_500 }, + }), + { kind: 'measured', tokens: 130 }, + ); + // No snapshot at all leaves the anchor standing. + assert.deepEqual( + resolveContextUsage({ latestRequestUsage: { kind: 'tokens', tokens: 120 } }), + { kind: 'measured', tokens: 120 }, + ); + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage: { kind: 'tokens', tokens: 120, at: 1_500 }, + live: { usageTokens: 130, completedAt: 1_500 }, + }), + { kind: 'measured', tokens: 130 }, + ); + // The snapshot can still vouch when the transcript established nothing. + assert.deepEqual( + resolveContextUsage({ latestRequestUsage: undefined, live: { usageTokens: 130 } }), + { kind: 'measured', tokens: 130 }, + ); + assert.deepEqual(resolveContextUsage({ latestRequestUsage: undefined }), { + kind: 'unavailable', + }); +}); + +test('a boundary supersedes the snapshot it landed after', () => { + // The manual `/compact` case: the snapshot still describes the pre-compaction + // prompt, so the gauge says unknown rather than holding that figure. + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage: { kind: 'compacted', at: 2_000 }, + live: { usageTokens: 90_000, completedAt: 1_000 }, + }), + { kind: 'stale', reason: 'compaction' }, + ); + assert.deepEqual( + resolveContextUsage({ latestRequestUsage: { kind: 'compacted', at: 2_000 } }), + { kind: 'stale', reason: 'compaction' }, + ); + // An untimed snapshot cannot be shown to be newer than a boundary, and a + // guess in that position is the precise-looking lie this state exists to + // refuse. + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage: { kind: 'compacted', at: 2_000 }, + live: { usageTokens: 90_000 }, + }), + { kind: 'stale', reason: 'compaction' }, + ); +}); + +test('a snapshot newer than the boundary is the post-compaction reading', () => { + // A mid-turn compaction is followed by steps that really do measure the smaller + // prompt, so the gauge recovers without waiting for the turn to end. + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage: { kind: 'compacted', at: 2_000 }, + live: { usageTokens: 30_000, completedAt: 2_500 }, + }), + { kind: 'measured', tokens: 30_000 }, + ); +}); + + +test('a selected live measurement carries only its own metered window', () => { + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage: { kind: 'tokens', tokens: 120 }, + live: { usageTokens: 130, contextWindow: 1_000, completedAt: 1_500 }, + }), + { kind: 'measured', tokens: 130, meteredWindow: 1_000 }, + ); + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage: { kind: 'compacted', at: 2_000 }, + live: { usageTokens: 130, contextWindow: 1_000, completedAt: 1_500 }, + }), + { kind: 'stale', reason: 'compaction' }, + ); +}); + +test('equal or missing boundary times cannot establish a post-compaction measurement', () => { + for (const at of [undefined, 2_000]) { + assert.deepEqual( + resolveContextUsage({ + latestRequestUsage: { kind: 'compacted', at }, + live: { usageTokens: 130, completedAt: 2_000 }, + }), + { kind: 'stale', reason: 'compaction' }, + ); + } }); diff --git a/apps/desktop/src/main/__tests__/live-context-usage.test.ts b/apps/desktop/src/main/__tests__/live-context-usage.test.ts index 774ffb2f7a..1c8f542853 100644 --- a/apps/desktop/src/main/__tests__/live-context-usage.test.ts +++ b/apps/desktop/src/main/__tests__/live-context-usage.test.ts @@ -83,6 +83,7 @@ function scriptedQuery() { describe('liveContextUsageFromDiagnostics', () => { it('maps a matching snapshot onto the gauge, window included', () => { assert.deepEqual(liveContextUsageFromDiagnostics(available(), ROUTE), { + completedAt: 1, usageTokens: 79_436, contextWindow: 128_000, }); @@ -113,7 +114,7 @@ describe('liveContextUsageFromDiagnostics', () => { it('stands alone without a window', () => { assert.deepEqual( liveContextUsageFromDiagnostics(available({ contextWindow: undefined }), ROUTE), - { usageTokens: 79_436 }, + { completedAt: 1, usageTokens: 79_436 }, ); }); }); @@ -136,7 +137,7 @@ describe('createLiveContextUsageTracker', () => { await Promise.resolve(); // The leading `undefined` is the aim itself: whatever stood on screen // before cannot answer for this target, so it clears before the read. - assert.deepEqual(seen, [undefined, { usageTokens: 79_436, contextWindow: 128_000 }]); + assert.deepEqual(seen, [undefined, { completedAt: 1, usageTokens: 79_436, contextWindow: 128_000 }]); tracker.dispose(); }); @@ -182,12 +183,12 @@ describe('createLiveContextUsageTracker', () => { assert.equal(query.pending.length, 1); timer.fire(); assert.equal(query.pending.length, 2); - query.pending[1]!.resolve(available({ inputTokens: 52_000 })); + query.pending[1]!.resolve(available({ inputTokens: 52_000, completedAt: 2 })); await Promise.resolve(); assert.deepEqual(seen, [ undefined, - { usageTokens: 40_000, contextWindow: 128_000 }, - { usageTokens: 52_000, contextWindow: 128_000 }, + { completedAt: 1, usageTokens: 40_000, contextWindow: 128_000 }, + { completedAt: 2, usageTokens: 52_000, contextWindow: 128_000 }, ]); tracker.dispose(); }); @@ -231,7 +232,7 @@ describe('createLiveContextUsageTracker', () => { await Promise.resolve(); query.pending[0]!.resolve(available({ inputTokens: 10_000 })); await Promise.resolve(); - assert.deepEqual(seen, [undefined, { usageTokens: 60_000, contextWindow: 128_000 }]); + assert.deepEqual(seen, [undefined, { completedAt: 1, usageTokens: 60_000, contextWindow: 128_000 }]); tracker.dispose(); }); @@ -258,7 +259,7 @@ describe('createLiveContextUsageTracker', () => { timer.fire(); query.pending[1]!.resolve(available({ inputTokens: 60_000 })); await Promise.resolve(); - assert.deepEqual(seen, [undefined, { usageTokens: 60_000, contextWindow: 128_000 }]); + assert.deepEqual(seen, [undefined, { completedAt: 1, usageTokens: 60_000, contextWindow: 128_000 }]); tracker.dispose(); }); @@ -281,7 +282,7 @@ describe('createLiveContextUsageTracker', () => { query.pending[1]!.reject(new Error('host not ready')); await Promise.resolve(); await Promise.resolve(); - assert.deepEqual(seen, [undefined, { usageTokens: 79_436, contextWindow: 128_000 }]); + assert.deepEqual(seen, [undefined, { completedAt: 1, usageTokens: 79_436, contextWindow: 128_000 }]); tracker.dispose(); }); @@ -303,14 +304,14 @@ describe('createLiveContextUsageTracker', () => { // Switching sessions makes the standing number unanswerable: it must // leave the screen BEFORE the new target's first read lands… tracker.setTarget({ sessionId: 's2', route: ROUTE }); - assert.deepEqual(seen, [undefined, { usageTokens: 79_436, contextWindow: 128_000 }, undefined]); + assert.deepEqual(seen, [undefined, { completedAt: 1, usageTokens: 79_436, contextWindow: 128_000 }, undefined]); // …and a rejected first read on the new target keeps it cleared, rather // than pinning the previous session's number in place indefinitely. query.pending[1]!.reject(new Error('host not ready')); await Promise.resolve(); await Promise.resolve(); - assert.deepEqual(seen, [undefined, { usageTokens: 79_436, contextWindow: 128_000 }, undefined]); + assert.deepEqual(seen, [undefined, { completedAt: 1, usageTokens: 79_436, contextWindow: 128_000 }, undefined]); tracker.dispose(); }); @@ -337,7 +338,7 @@ describe('createLiveContextUsageTracker', () => { query.pending[1]!.reject(new Error('host not ready')); await Promise.resolve(); await Promise.resolve(); - assert.deepEqual(seen, [undefined, { usageTokens: 79_436, contextWindow: 128_000 }]); + assert.deepEqual(seen, [undefined, { completedAt: 1, usageTokens: 79_436, contextWindow: 128_000 }]); tracker.dispose(); }); @@ -382,7 +383,7 @@ describe('createLiveContextUsageTracker', () => { await Promise.resolve(); // Aiming, then leaving s1 clears its (never-landed) reading, then s2's // lands; the stale s1 read resolving late must not overwrite it. - assert.deepEqual(seen, [undefined, undefined, { usageTokens: 5_000, contextWindow: 128_000 }]); + assert.deepEqual(seen, [undefined, undefined, { completedAt: 1, usageTokens: 5_000, contextWindow: 128_000 }]); tracker.dispose(); }); @@ -408,7 +409,7 @@ describe('createLiveContextUsageTracker', () => { await Promise.resolve(); assert.deepEqual(seen, [ undefined, - { usageTokens: 79_436, contextWindow: 128_000 }, + { completedAt: 1, usageTokens: 79_436, contextWindow: 128_000 }, undefined, undefined, ]); diff --git a/apps/desktop/src/renderer/application/contracts/session-inspector/latest-request-usage.ts b/apps/desktop/src/renderer/application/contracts/session-inspector/latest-request-usage.ts index ce32b827fe..fc9d46f4b3 100644 --- a/apps/desktop/src/renderer/application/contracts/session-inspector/latest-request-usage.ts +++ b/apps/desktop/src/renderer/application/contracts/session-inspector/latest-request-usage.ts @@ -17,50 +17,188 @@ * under the License. */ -/** - * The session's latest provider-counted request, or nothing. - * - * A token count belongs to one request on one route: it is a number in that - * model's tokenizer, and it is only the session's latest if nothing newer - * exists. The runtime enforces both when it reads an anchor back, refusing one - * whose run header names another model or connection. A control that shows the - * number has to enforce the same two facts or it will display a precise-looking - * figure about a request the user is not making — model A's tokens against - * model B's window, or a historical range's usage presented as current. - * - * So this refuses rather than approximates, and the three refusals are the - * three normal states that break the pairing: - * - * - the loaded transcript range is not the session tail, so a newer request may - * exist that this range cannot see; - * - the newest usage row carries no anchor, which is what manual `/compact` - * writes, so the scan continues past it exactly as the runtime's does; - * - the anchor names a different route than the active one, or names none at - * all because it was written before anchors carried their route. - */ +import type { ContextUsageReading } from '@maka/ui'; +import type { LiveContextUsage } from './live-context-usage.js'; + export interface LatestRequestUsageAnchor { inputTokens: number; outputTokens?: number; + completedAt?: number; modelId?: string; connectionId?: string; } +export interface LatestRequestUsageRow { + readonly type: string; + readonly ts?: number; + readonly kind?: string; + readonly turnId?: string; + readonly lastRequestAnchor?: LatestRequestUsageAnchor; +} + +export type LatestRequestUsage = + | { readonly kind: 'tokens'; readonly tokens: number; readonly at?: number } + | { readonly kind: 'compacted'; readonly at?: number } + | undefined; + +/** + * Read the newest route-matching measurement or compaction from the session tail. + * Anchorless usage rows (including manual compaction usage) carry no measurement. + * A compaction invalidates earlier measurements until a later request settles. + * + * Ledger position decides the scan order, but the candidate is arbitrated by + * event time: settlement persists the usage row AFTER the compaction notes even + * when the anchored request completed BEFORE the compaction — its retry never + * finished (#5547) — so a boundary row behind the newest anchored row can still + * supersede it when the compaction's apply time postdates the anchor's completion. + */ export function selectLatestRequestUsage( - messages: readonly { type: string; lastRequestAnchor?: LatestRequestUsageAnchor }[], + messages: readonly LatestRequestUsageRow[], model: string | undefined, route: { llmConnectionId?: string } | undefined, -): number | undefined { +): LatestRequestUsage { const connectionId = route?.llmConnectionId; - if (model === undefined || connectionId === undefined) return undefined; + // The newest anchored usage row, held while the scan behind it looks for a + // boundary that postdates its completion. A second anchored row settles it: + // under ordered writes every boundary behind that row is strictly older. + let pendingTokens: + | { + readonly reading: { + readonly kind: 'tokens'; + readonly tokens: number; + readonly at?: number; + }; + readonly completedAt?: number; + } + | undefined; for (let index = messages.length - 1; index >= 0; index -= 1) { const message = messages[index]; + if (message?.type === 'system_note' && message.kind === 'context_compaction_applied') { + // Written the moment the compaction is applied, so its row time IS the + // boundary time — mid-turn compactions land it mid-turn, not at settlement. + if (!pendingTokens) { + return { kind: 'compacted', ...(message.ts !== undefined ? { at: message.ts } : {}) }; + } + return orderTokensAgainstBoundary(pendingTokens, message.ts); + } + if (message?.type === 'system_note' && message.kind === 'context_compacted') { + // The settlement-time display row: the compaction it describes was applied + // earlier in the same turn and recorded its own boundary row. + const appliedAt = latestCompactionAppliedAt(messages, index, message.turnId); + const at = appliedAt ?? message.ts; + if (!pendingTokens) { + return { kind: 'compacted', ...(at !== undefined ? { at } : {}) }; + } + return orderTokensAgainstBoundary(pendingTokens, at); + } if (message?.type !== 'token_usage') continue; const anchor = message.lastRequestAnchor; if (!anchor) continue; + if (pendingTokens) return pendingTokens.reading; + if (model === undefined || connectionId === undefined) return undefined; if (anchor.modelId !== model || anchor.connectionId !== connectionId) return undefined; if (!Number.isFinite(anchor.inputTokens) || anchor.inputTokens <= 0) return undefined; const output = Number.isFinite(anchor.outputTokens ?? 0) ? Math.max(0, anchor.outputTokens ?? 0) : 0; - return anchor.inputTokens + output; + pendingTokens = { + completedAt: anchor.completedAt, + reading: { + kind: 'tokens', + tokens: anchor.inputTokens + output, + // The row is persisted after request settlement (and sometimes after a + // compaction note). Its write time cannot order its own snapshot. + ...(anchor.completedAt !== undefined ? { at: anchor.completedAt } : {}), + }, + }; + } + return pendingTokens?.reading; +} + +/** + * Arbitration between the position-newest anchored usage row and a boundary + * row found behind it. The compaction supersedes the measurement only when its + * apply time provably postdates the anchored request's completion; a missing + * completion or boundary time cannot establish that order, so the candidate + * stands. + */ +function orderTokensAgainstBoundary( + pendingTokens: { + readonly reading: { readonly kind: 'tokens'; readonly tokens: number; readonly at?: number }; + readonly completedAt?: number; + }, + boundaryAt: number | undefined, +): LatestRequestUsage { + if ( + pendingTokens.completedAt !== undefined && + boundaryAt !== undefined && + boundaryAt >= pendingTokens.completedAt + ) { + return { kind: 'compacted', at: boundaryAt }; + } + return pendingTokens.reading; +} + +/** + * Find the apply-time boundary behind a settlement `context_compacted` row. + * The display note is written when the turn settles while the compaction it + * describes was applied mid-turn; between the two rows sit only that turn's + * post-compaction output — the turn's own usage row lands after the note — so the + * first relevant row behind the note is the matching apply row when one was + * recorded. A different turn's apply row, an older display note, a usage + * row, or the head of the log all mean the note's own write time is all + * there is. + */ +function latestCompactionAppliedAt( + messages: readonly LatestRequestUsageRow[], + noteIndex: number, + turnId: string | undefined, +): number | undefined { + for (let index = noteIndex - 1; index >= 0; index -= 1) { + const message = messages[index]; + if (message?.type === 'token_usage') return undefined; + if (message?.type !== 'system_note') continue; + if (message.kind === 'context_compaction_applied') { + return message.turnId === turnId ? message.ts : undefined; + } + if (message.kind === 'context_compacted') return undefined; } return undefined; } + +/** + * Resolve the context-usage reading the UI can display from the live Turn's + * latest request snapshot and the durable transcript's token-usage anchors + * and compaction notes. Choose the newest reading whose relative order can + * be established; after a compaction, only a measurement proven to have + * completed later may be shown, otherwise report the usage as stale. + */ +export function resolveContextUsage(input: { + readonly latestRequestUsage: LatestRequestUsage; + readonly live?: LiveContextUsage; +}): ContextUsageReading { + const { latestRequestUsage, live } = input; + if ( + latestRequestUsage?.kind === 'compacted' && + (live?.completedAt === undefined || + latestRequestUsage.at === undefined || + latestRequestUsage.at >= live.completedAt) + ) { + return { kind: 'stale', reason: 'compaction' }; + } + if ( + latestRequestUsage?.kind === 'tokens' && + latestRequestUsage.at !== undefined && + (live?.completedAt === undefined || latestRequestUsage.at > live.completedAt) + ) { + return { kind: 'measured', tokens: latestRequestUsage.tokens }; + } + if (live) { + return { + kind: 'measured', + tokens: live.usageTokens, + ...(live.contextWindow !== undefined ? { meteredWindow: live.contextWindow } : {}), + }; + } + if (latestRequestUsage?.kind === 'tokens') + return { kind: 'measured', tokens: latestRequestUsage.tokens }; + return { kind: 'unavailable' }; +} diff --git a/apps/desktop/src/renderer/application/contracts/session-inspector/live-context-usage.ts b/apps/desktop/src/renderer/application/contracts/session-inspector/live-context-usage.ts index e94a6f08a2..3b912d2dea 100644 --- a/apps/desktop/src/renderer/application/contracts/session-inspector/live-context-usage.ts +++ b/apps/desktop/src/renderer/application/contracts/session-inspector/live-context-usage.ts @@ -44,6 +44,14 @@ export interface LiveContextUsage { readonly usageTokens: number; /** The window the request was metered against, frozen at call time. */ readonly contextWindow?: number; + /** + * When that request settled, on the Host's clock — the same clock the + * session's own transcript rows carry, so a reader can tell whether this + * snapshot predates a compaction boundary it already knows about. Without + * it, a snapshot that a compaction has replaced is indistinguishable from one + * taken after it. + */ + readonly completedAt?: number; } /** @@ -56,6 +64,10 @@ export interface LiveContextUsage { * numerator by the same row's denominator exactly as the inspector's bar * does — a window from the live catalog could disagree with the metered * request, while a user-declared override still wins by design. + * + * The settlement time rides along for the same reason the window does: it + * belongs to this request alone, and a reader comparing this reading against a + * compaction boundary has no other row to get it from. */ export function liveContextUsageFromDiagnostics( diagnostics: ContextDiagnosticsResult | undefined, @@ -75,6 +87,7 @@ export function liveContextUsageFromDiagnostics( ...(diagnostics.contextWindow !== undefined ? { contextWindow: diagnostics.contextWindow } : {}), + completedAt: diagnostics.completedAt, }; } diff --git a/apps/desktop/src/renderer/chat-composer-region.tsx b/apps/desktop/src/renderer/chat-composer-region.tsx index b8d61e0da1..e7ae25514a 100644 --- a/apps/desktop/src/renderer/chat-composer-region.tsx +++ b/apps/desktop/src/renderer/chat-composer-region.tsx @@ -31,7 +31,11 @@ import { } from '@maka/ui'; import { StagedComposer, type ComposerStagingProp } from './features/conversation/index.js'; import type { ComposerHandle } from '@maka/ui'; -export { selectLatestRequestUsage } from './application/contracts/session-inspector/latest-request-usage.js'; +import { + resolveContextUsage, + type LatestRequestUsage, +} from './application/contracts/session-inspector/latest-request-usage.js'; +import type { LiveContextUsage } from './application/contracts/session-inspector/live-context-usage.js'; import { useComposerMentionsContext } from './composer-mentions.js'; import type { GuestComposerProjection } from './features/session-collaboration/index.js'; import { @@ -124,14 +128,14 @@ interface ChatComposerRegionProps boundaryUnreadableNotice?: BoundaryUnreadableNotice; /** * Tokens the provider counted for the session's latest request on the active - * route, or nothing when that cannot be established. Resolved by the owner, - * which knows the transcript range and the route; this control never derives - * it from the rendered slice. This is the per-turn anchor: it moves when a - * turn's usage record lands. `LiveContextUsageProbe` overlays the - * per-settled-request snapshot (#4717) whenever that snapshot can vouch for - * the same route, and this value is the fallback when it cannot. + * route, or nothing when that cannot be established, or a compaction + * boundary that superseded it. Resolved by the owner, which knows the + * transcript range and the route; this control never derives it from the + * rendered slice. `LiveContextUsageProbe` overlays the per-settled-request + * snapshot (#4717) whenever that snapshot is the newer answer to the same + * question, and this value is the fallback when it is not. */ - latestRequestUsageTokens?: number; + latestRequestUsage?: LatestRequestUsage; onOpenContextUsage(): void; /** * The live overlay for the gauge (#4717), injected rather than imported: @@ -151,7 +155,7 @@ interface ChatComposerRegionProps * ceiling. */ children: ( - usage: { readonly usageTokens: number; readonly contextWindow?: number } | undefined, + usage: LiveContextUsage | undefined, ) => ReactNode; }>; canStageContext: boolean; @@ -176,7 +180,7 @@ export function ChatComposerRegion({ respondToUserForm, stop, boundaryUnreadableNotice, - latestRequestUsageTokens, + latestRequestUsage, onOpenContextUsage, LiveContextUsageProbe, canStageContext, @@ -200,14 +204,6 @@ export function ChatComposerRegion({ choice.model === composerRest.activeModel, ) : undefined; - const contextUsage = activeId - ? { - usageTokens: latestRequestUsageTokens, - declaredContextWindow: activeModelChoice?.declaredContextWindow, - metadataContextWindow: activeModelChoice?.contextWindow, - onOpen: onOpenContextUsage, - } - : undefined; const previousNewTaskDraftKey = useRef(newTaskDraftKey); useLayoutEffect(() => { const previous = previousNewTaskDraftKey.current; @@ -269,49 +265,57 @@ export function ChatComposerRegion({ // — when mounted — can feed it the per-settled-request snapshot (#4717), and // the anchor prop remains the reading it falls back to. const renderComposer = ( - liveContextUsage: { readonly usageTokens: number; readonly contextWindow?: number } | undefined, - ) => ( - - {(goalProjection) => ( - - ); + liveContextUsage: LiveContextUsage | undefined, + ) => { + // One question, two answers, and a compaction can make the finer one stale: the + // snapshot wins when it landed after the boundary, and the boundary wins + // when it did not. + const reading = resolveContextUsage({ latestRequestUsage, live: liveContextUsage }); + const contextUsage = activeId + ? { + reading, + declaredContextWindow: activeModelChoice?.declaredContextWindow, + metadataContextWindow: activeModelChoice?.contextWindow, + onOpen: onOpenContextUsage, + } + : undefined; + return ( + + {(goalProjection) => ( + + ); + }; return ( <> diff --git a/apps/desktop/src/renderer/features/conversation/model/conversation-workspace.ts b/apps/desktop/src/renderer/features/conversation/model/conversation-workspace.ts index 258ac69b7b..e16a488f77 100644 --- a/apps/desktop/src/renderer/features/conversation/model/conversation-workspace.ts +++ b/apps/desktop/src/renderer/features/conversation/model/conversation-workspace.ts @@ -124,8 +124,18 @@ export function createConversationWorkspace(catalog: SessionCatalogController, s ui, activeIdRef, transcriptRangeRef, bootstrapSelectionLease, commands, publishedSession: Object.freeze({ get current() { return activeIdRef.current; } }), messages: reader((value) => value.messages), - usage: (model: string | undefined, connectionId: string | undefined) => reader((value) => - selectLatestRequestUsage(value.messages, model, { llmConnectionId: connectionId })), + usage: (model: string | undefined, connectionId: string | undefined) => { + let messages: readonly StoredMessage[] | undefined; + let usage: ReturnType; + return reader((value) => { + // Keep the external-store snapshot stable until its transcript changes. + if (messages !== value.messages) { + messages = value.messages; + usage = selectLatestRequestUsage(messages, model, { llmConnectionId: connectionId }); + } + return usage; + }); + }, publication: reader((value) => value), target: reader((value) => value.sessionId), chrome: reader((value) => ({ diff --git a/apps/desktop/src/renderer/features/conversation/ui/conversation-readers.tsx b/apps/desktop/src/renderer/features/conversation/ui/conversation-readers.tsx index f63ce9e5ba..2606164a01 100644 --- a/apps/desktop/src/renderer/features/conversation/ui/conversation-readers.tsx +++ b/apps/desktop/src/renderer/features/conversation/ui/conversation-readers.tsx @@ -34,6 +34,7 @@ import { useAppShellTurnPresentation } from '../../../application/contracts/turn import { getShellCopy } from '../../../locales/shell-copy.js'; import { composerTurnGates, desktopComposerSlashCommands } from '../model/composer-turn-gates.js'; import { executorComposerProps } from '../model/executor-composer.js'; +import type { LatestRequestUsage } from '../../../application/contracts/session-inspector/latest-request-usage.js'; import { liveTurnFlags } from '../model/live-turn-flags.js'; /** The displayed Session's Turn, read at chrome frequency. */ @@ -121,7 +122,7 @@ type TurnGatedProps = ReturnType & ReturnType['composer']; - processing: boolean; pendingMessages: ChatProps['transientMessages']; latestRequestUsageTokens?: number; + processing: boolean; pendingMessages: ChatProps['transientMessages']; latestRequestUsage?: LatestRequestUsage; /** The owner Session's open interaction; the Composer slot answers it. */ activeInteraction: ComposerInteraction | undefined; queuedMessages: UiComposerProps['queuedMessages']; @@ -168,7 +169,7 @@ export function ConversationComposerRegion

( const { selection: executor, ...executorInput } = executorComposer; const view = useSyncExternalStore(workspace.composer.subscribe, workspace.composer.getSnapshot); const usage = useMemo(() => workspace.usage(usageModel, usageRoute?.llmConnectionId), [workspace, usageModel, usageRoute?.llmConnectionId]); - const latestRequestUsageTokens = useSyncExternalStore(usage.subscribe, usage.getSnapshot); + const latestRequestUsage = useSyncExternalStore(usage.subscribe, usage.getSnapshot); const stopPending = useSessionUiRead(workspace.ui.reads, 'stop', activeId); const draft = submission.revisionDraft; const editing = draft !== null && activeId === draft.draftSessionId; @@ -214,7 +215,7 @@ export function ConversationComposerRegion

( : {}), processing: view.transientMessages.length > 0, pendingMessages: view.transientMessages, - latestRequestUsageTokens, + latestRequestUsage, }; return createElement(surface, { ...presentation, ...owned } as unknown as P); } diff --git a/apps/desktop/src/renderer/features/workhub/ui/workhub-root.tsx b/apps/desktop/src/renderer/features/workhub/ui/workhub-root.tsx index 3852f43956..6585164643 100644 --- a/apps/desktop/src/renderer/features/workhub/ui/workhub-root.tsx +++ b/apps/desktop/src/renderer/features/workhub/ui/workhub-root.tsx @@ -22,7 +22,10 @@ import { ChatSurfaceLayout, UserQuestionPrompt, MakaWordmark, useUiLocale, type import { Button, IconButton } from '@astryxdesign/core'; import { ChevronDown, PictureInPicture2, Undo2, X } from '@maka/ui/icons'; import { useLiveContextUsage } from '../../../application/contracts/session-inspector/use-live-context-usage.js'; -import { selectLatestRequestUsage } from '../../../application/contracts/session-inspector/latest-request-usage.js'; +import { + resolveContextUsage, + selectLatestRequestUsage, +} from '../../../application/contracts/session-inspector/latest-request-usage.js'; import { WorkHubProgressCard } from './workhub-progress-card.js'; import { WorkHubComposer } from './workhub-composer.js'; import type { RestoredDraftContent } from '../../../application/contracts/transient-message-projection.js'; @@ -101,10 +104,19 @@ export function WorkHubRoot() { ); const thinkingLevels = newWorkModelChoice?.thinkingLevels ?? []; const liveContextUsage = useLiveContextUsage({ inspector: services.inspector, sessionId: controller.sessionId, model: session?.model, providerType: coordinationModelChoice?.providerType }); + const contextUsageReading = useMemo( + () => + resolveContextUsage({ + latestRequestUsage: selectLatestRequestUsage(transcript.messages, session?.model, session), + live: liveContextUsage, + }), + [transcript.messages, session, liveContextUsage], + ); const thinkingLevel = controller.newWorkDefaults.thinkingLevel && thinkingLevels.includes(controller.newWorkDefaults.thinkingLevel) ? controller.newWorkDefaults.thinkingLevel : undefined; + const locale = useUiLocale(); const t = workHubLiveCopy[locale]; const shortcutLabel = navigator.platform.toLowerCase().includes('mac') ? '⌘⇧K' : 'Ctrl+Shift+K'; @@ -369,9 +381,8 @@ export function WorkHubRoot() { : undefined} modelSwitchAvailability={controller.configuringModel ? { available: false, pending: true, reason: 'pending' } : undefined} contextUsage={session ? { - usageTokens: liveContextUsage?.usageTokens ?? selectLatestRequestUsage(transcript.messages, session.model, session), + reading: contextUsageReading, declaredContextWindow: coordinationModelChoice?.declaredContextWindow, - meteredContextWindow: liveContextUsage?.contextWindow, metadataContextWindow: coordinationModelChoice?.contextWindow, onOpen: () => call(services.presentation.openUsage()), } : undefined} diff --git a/apps/desktop/stories/app-shell.stories.tsx b/apps/desktop/stories/app-shell.stories.tsx index 84b3f94719..7d176e67fb 100644 --- a/apps/desktop/stories/app-shell.stories.tsx +++ b/apps/desktop/stories/app-shell.stories.tsx @@ -3018,7 +3018,7 @@ export const ReaderScrolledUpIsNotPulledBack: Story = { turns={12} composer={{ contextUsage: { - usageTokens: 37_000, + reading: { kind: 'measured', tokens: 37_000 }, declaredContextWindow: 100_000, onOpen: noop, }, @@ -4428,7 +4428,7 @@ export const NarrowComposerFooter: Story = { planModeActive: true, orchestrationMode: 'swarm', contextUsage: { - usageTokens: 100_000, + reading: { kind: 'measured', tokens: 100_000 }, declaredContextWindow: 100_000, onOpen: noop, }, diff --git a/packages/core/src/__tests__/usage-record-last-request-anchor.test.ts b/packages/core/src/__tests__/usage-record-last-request-anchor.test.ts index e37c5af83e..89b5cf3234 100644 --- a/packages/core/src/__tests__/usage-record-last-request-anchor.test.ts +++ b/packages/core/src/__tests__/usage-record-last-request-anchor.test.ts @@ -27,6 +27,8 @@ const usage = { input: 370, output: 60 }; test('a last-request anchor accepts the new usage shape and retired payload key', () => { assert.equal(isLastRequestAnchor({ inputTokens: 120, outputTokens: 30 }), true); assert.equal(isLastRequestAnchor({ inputTokens: 120 }), true); + assert.equal(isLastRequestAnchor({ inputTokens: 120, completedAt: 1_000 }), true); + assert.equal(isLastRequestAnchor({ inputTokens: 120, completedAt: -1 }), false); assert.equal(isLastRequestAnchor({ inputTokens: 120, payloadChars: 4_000 }), true); assert.equal(isLastRequestAnchor({ payloadChars: 4_000 }), false); assert.equal(isLastRequestAnchor({ inputTokens: 0, payloadChars: 4_000 }), false); diff --git a/packages/core/src/session.ts b/packages/core/src/session.ts index fd18305681..4afa07ffaf 100644 --- a/packages/core/src/session.ts +++ b/packages/core/src/session.ts @@ -835,7 +835,10 @@ export function userFacingText(message: Pick()( ['inputTokens'], - ['outputTokens', 'modelId', 'connectionId'], + ['outputTokens', 'completedAt', 'modelId', 'connectionId'], ['payloadChars'], ); @@ -329,6 +331,8 @@ export function isLastRequestAnchor(value: unknown): value is LastRequestAnchor value.inputTokens > 0 && (value.outputTokens === undefined || (isFiniteNumber(value.outputTokens) && value.outputTokens >= 0)) && + (value.completedAt === undefined || + (isFiniteNumber(value.completedAt) && value.completedAt >= 0)) && (value.payloadChars === undefined || isFiniteNumber(value.payloadChars)) ); } diff --git a/packages/runtime-host/protocol-compatible-changes/epoch-note-compaction-kind.json b/packages/runtime-host/protocol-compatible-changes/epoch-note-compaction-kind.json new file mode 100644 index 0000000000..3a78fb729e --- /dev/null +++ b/packages/runtime-host/protocol-compatible-changes/epoch-note-compaction-kind.json @@ -0,0 +1,5 @@ +{ + "epoch": 205, + "files": ["packages/runtime-host/src/protocol/index.ts"], + "reason": "Extends the epoch-205 comment to also mention context_compaction_applied system notes; the epoch value and every wire shape are unchanged." +} diff --git a/packages/runtime-host/src/protocol/index.ts b/packages/runtime-host/src/protocol/index.ts index 712f0c82c0..319a890020 100644 --- a/packages/runtime-host/src/protocol/index.ts +++ b/packages/runtime-host/src/protocol/index.ts @@ -104,7 +104,11 @@ export const RUNTIME_HOST_REGISTRATION_SCHEMA_VERSION = 1 as const; export const RUNTIME_HOST_PROTOCOL_VERSION = 0 as const; // Increment when the same protocol version no longer guarantees safe Client-Host // interoperability. Mismatches are rejected before domain commands are admitted. -export const RUNTIME_HOST_COMPATIBILITY_EPOCH = 204 as const; +export const RUNTIME_HOST_COMPATIBILITY_EPOCH = 205 as const; +// 205: token_usage.lastRequestAnchor may carry completedAt, the provider +// request settlement time, and transcripts may carry +// context_compaction_applied system notes. Older peers reject both in the +// closed transcript shape. // 204: Executor catalogs and Session configuration carry opaque mode IDs; // catalog queries may request a provider refresh. Older peers reject these fields. // 202: `session.remove.preview` takes a bounded list of Sessions and reports the diff --git a/packages/runtime/src/__tests__/mid-turn-capacity-backend.test.ts b/packages/runtime/src/__tests__/mid-turn-capacity-backend.test.ts index add55fad4b..257a1b041d 100644 --- a/packages/runtime/src/__tests__/mid-turn-capacity-backend.test.ts +++ b/packages/runtime/src/__tests__/mid-turn-capacity-backend.test.ts @@ -1722,6 +1722,7 @@ describe('the shipped runtime default drives the proactive long-turn journey (is const fixture = buildFixture({ contextWindow: 1_000_000, finalAtSecondCall: true, + meteredSummarizer: true, }); await runFixtureTurn(fixture); @@ -1737,10 +1738,12 @@ describe('the shipped runtime default drives the proactive long-turn journey (is // Two steps: 100 + 120 reported input, and the send sum is both. assert.equal(usage?.input, 220); const anchor = usage?.lastRequestAnchor as - | { inputTokens: number; outputTokens?: number } + | { inputTokens: number; outputTokens?: number; completedAt?: number } | undefined; assert.equal(anchor?.inputTokens, 120); assert.equal(anchor?.outputTokens, 10); + const lastMainAttempt = fixture.modelCalls.filter((call) => call.callKind === 'main').at(-1); + assert.equal(anchor?.completedAt, lastMainAttempt?.completedAt); }); test('an anchor is discarded unless its invocation proves it came from this model', async () => { @@ -1811,6 +1814,139 @@ describe('the shipped runtime default drives the proactive long-turn journey (is assert.equal(fold?.phase, 'pre_turn'); }); + test('a compaction boundary supersedes an anchor whose request predates the compaction (#5547)', async () => { + // The failed-send ledger shape: the settlement usage row lands after the + // apply row, but its anchor still describes the pre-compaction request — the + // retry never completed. Such an anchor must not seed the next send's + // baseline: it measures context the compaction already replaced. + for (const [boundaryTs, anchorCompletedAt, folds] of [ + [1_800_000_001_000, 1_800_000_000_500, false], + [1_800_000_000_100, 1_800_000_000_500, true], + ] as const) { + const fixture = buildFixture({ + priorChars: 2_000, + contextWindow: 20_000, + finalAtSecondCall: true, + extraPriorEvents: [ + { + ...runtimeTextEvent(`prior-applied-${boundaryTs}`, 'turn-0', 'model', ''), + runId: 'run-0', + invocationId: 'run-0', + role: 'system' as const, + author: 'system' as const, + ts: boundaryTs, + content: { + kind: 'system_note' as const, + note: 'context_compaction_applied', + }, + }, + priorUsageEvent({ + inputTokens: 30_000, + outputTokens: 10, + completedAt: anchorCompletedAt, + }), + ], + priorInvocations: [priorRunInvocation()], + }); + await runFixtureTurn(fixture); + + assert.equal(fixture.recorded.length, folds ? 1 : 0); + } + }); + + test('a settlement compaction note delegates its boundary to the apply row', async () => { + // The display note is written when the turn settles — after every + // request of its turn — so ordering an anchor against the note's own + // row time discards every post-compaction measurement. The apply row carries + // the real boundary: an anchor completed after the compaction still seeds + // the next send's baseline, and one completed before it does not. + for (const [appliedTs, anchorCompletedAt, folds] of [ + // Post-compaction anchor: applied at T2, anchor completed at T4, note + // settled at T5 — the anchor survives and the over-window seed + // compacts step 0. + [1_800_000_001_000, 1_800_000_004_000, true], + // Stale anchor: applied at T2 but the anchor completed at T1 — the + // retry never finished, so nothing post-compaction is measured. + [1_800_000_001_000, 1_800_000_000_500, false], + ] as const) { + const fixture = buildFixture({ + priorChars: 2_000, + contextWindow: 20_000, + finalAtSecondCall: true, + extraPriorEvents: [ + { + ...runtimeTextEvent(`prior-applied-${appliedTs}`, 'turn-0', 'model', ''), + runId: 'run-0', + invocationId: 'run-0', + role: 'system' as const, + author: 'system' as const, + ts: appliedTs, + content: { + kind: 'system_note' as const, + note: 'context_compaction_applied', + }, + }, + { + ...runtimeTextEvent(`prior-note-${appliedTs}`, 'turn-0', 'model', ''), + runId: 'run-0', + invocationId: 'run-0', + role: 'system' as const, + author: 'system' as const, + ts: 1_800_000_005_000, + content: { + kind: 'system_note' as const, + note: 'context_compacted', + }, + }, + priorUsageEvent({ + inputTokens: 30_000, + outputTokens: 10, + completedAt: anchorCompletedAt, + }), + ], + priorInvocations: [priorRunInvocation()], + }); + await runFixtureTurn(fixture); + + assert.equal(fixture.recorded.length, folds ? 1 : 0); + } + }); + + test('a settlement note without an apply row is itself the boundary', async () => { + // Ledgers written before `context_compaction_applied` existed — or a + // failed apply-row write — leave the settlement note as the only + // boundary. Its time postdates every request of its turn, so no anchor + // can be shown to postdate the compaction and the conservative seed is none. + const fixture = buildFixture({ + priorChars: 2_000, + contextWindow: 20_000, + finalAtSecondCall: true, + extraPriorEvents: [ + { + ...runtimeTextEvent('prior-note-legacy', 'turn-0', 'model', ''), + runId: 'run-0', + invocationId: 'run-0', + role: 'system' as const, + author: 'system' as const, + ts: 1_800_000_005_000, + content: { + kind: 'system_note' as const, + note: 'context_compacted', + }, + }, + priorUsageEvent({ + inputTokens: 30_000, + outputTokens: 10, + completedAt: 1_800_000_004_000, + }), + ], + priorInvocations: [priorRunInvocation()], + }); + await runFixtureTurn(fixture); + + assert.equal(fixture.recorded.length, 0); + }); + test('the reserve is twice the last real reply, bounded, not the model output limit', async () => { // With the window declared at the provider's real size, an accepted // request can never exceed it on its own; the reply the next request must @@ -1887,6 +2023,7 @@ describe('the shipped runtime default drives the proactive long-turn journey (is function priorUsageEvent(lastRequestAnchor: { inputTokens: number; outputTokens?: number; + completedAt?: number; }): RuntimeEvent { return { ...runtimeTextEvent('prior-usage', 'turn-0', 'model', ''), diff --git a/packages/runtime/src/__tests__/overflow-reactive-recovery.test.ts b/packages/runtime/src/__tests__/overflow-reactive-recovery.test.ts index 31a0e812a3..309deb3b65 100644 --- a/packages/runtime/src/__tests__/overflow-reactive-recovery.test.ts +++ b/packages/runtime/src/__tests__/overflow-reactive-recovery.test.ts @@ -1903,6 +1903,55 @@ describe('reactive overflow recovery in the streaming backend', () => { assert.equal(fixture.llmCalls.at(-1)?.errorClass, 'context_overflow'); }); + test('a terminal retry after the compaction still settles usage with its completed anchor (#5547)', async () => { + // Step 1 completes; step 2 overflows; the recovery compaction lands; the retry + // overflows terminally. Settlement persists the completed steps' usage + // row — after the compaction notes — and its anchor still describes the + // PRE-compaction request. The fact is kept whole: `completedAt` is what lets a + // reader order the anchor against the apply-time boundary instead of + // trusting ledger position. + const fixture = buildReactiveFixture({ + script: ['tool', 'overflow', 'overflow'], + bigPriors: true, + }); + await runTurn(fixture); + + const usage = fixture.events.find((event) => event.type === 'token_usage') as + | { + type: 'token_usage'; + lastRequestAnchor?: { inputTokens?: number; completedAt?: number }; + } + | undefined; + assert.ok(usage); + assert.equal(typeof usage.lastRequestAnchor?.inputTokens, 'number'); + assert.equal(typeof usage.lastRequestAnchor?.completedAt, 'number'); + assert.equal( + fixture.messages.some( + (message) => + (message as { type?: string; kind?: string }).type === 'system_note' && + (message as { kind?: string }).kind === 'context_compaction_applied', + ), + true, + ); + }); + + test('a post-compaction retry that completes keeps its request anchor', async () => { + // The healthy counterpart: the retry completed after the compaction, so the + // settlement usage row anchors the NEXT turn's baseline on real post-compaction + // input. + const fixture = buildReactiveFixture({ + script: ['tool', 'overflow', 'done'], + bigPriors: true, + }); + await runTurn(fixture); + + const usage = fixture.events.find((event) => event.type === 'token_usage') as + | { type: 'token_usage'; lastRequestAnchor?: { inputTokens?: number } } + | undefined; + assert.ok(usage); + assert.equal(typeof usage.lastRequestAnchor?.inputTokens, 'number'); + }); + test('a proactive fold spends the step, so the same step does not fold again', async () => { // The declared window is crossed before the second request, so that // request is already the folded one when the provider rejects it. Reactive @@ -2190,7 +2239,8 @@ describe('reactive overflow recovery in the streaming backend', () => { const note = fixture.messages.find( (message): message is { type: 'system_note'; kind: string; data?: unknown } => - (message as { type?: string }).type === 'system_note', + (message as { type?: string }).type === 'system_note' && + (message as { kind?: string }).kind === 'context_window_suggestion', ); assert.equal(note?.kind, 'context_window_suggestion'); assert.deepEqual(note?.data, { @@ -2271,6 +2321,15 @@ describe('reactive overflow recovery in the streaming backend', () => { await runTurn(fixture); assert.equal(complete(fixture)?.stopReason, 'end_turn'); // The fold itself is noted (context_compacted); the window suggestion is not. + // The compaction landed mid-turn, so its apply-time boundary row exists too. + assert.equal( + fixture.messages.some( + (message) => + (message as { type?: string; kind?: string }).type === 'system_note' && + (message as { kind?: string }).kind === 'context_compaction_applied', + ), + true, + ); assert.equal( fixture.messages.some( (message) => diff --git a/packages/runtime/src/__tests__/provider-request-telemetry.test.ts b/packages/runtime/src/__tests__/provider-request-telemetry.test.ts index 72fdac6798..c20d51b1e5 100644 --- a/packages/runtime/src/__tests__/provider-request-telemetry.test.ts +++ b/packages/runtime/src/__tests__/provider-request-telemetry.test.ts @@ -265,6 +265,7 @@ describe('provider request tracker', () => { await drain(result.stream); assert.equal(attempts[0]?.contextWindow, 200_000); + assert.equal(tracker.latestCompletedMainRequestAt, attempts[0]?.completedAt); }); test('omits a non-positive request model context window', async () => { diff --git a/packages/runtime/src/__tests__/session-manager.test.ts b/packages/runtime/src/__tests__/session-manager.test.ts index 997cf7076b..4d3841a0a8 100644 --- a/packages/runtime/src/__tests__/session-manager.test.ts +++ b/packages/runtime/src/__tests__/session-manager.test.ts @@ -4180,6 +4180,14 @@ describe('SessionManager manual compaction and quiescent session changes', () => // The kernel writes the note on the compaction turn itself, so the row // appears the moment compaction ends — not one send later. assert.strictEqual(notes.length, 1); + // The apply-time boundary row lands beside it, on the same turn. + const applied = messages.filter( + (message) => + message.type === 'system_note' && + message.turnId === 'turn-compact' && + message.kind === 'context_compaction_applied', + ); + assert.strictEqual(applied.length, 1); // And no duplicate failed-open note on a successful compaction. const failed = messages.filter( (message) => diff --git a/packages/runtime/src/ai-sdk-compaction.ts b/packages/runtime/src/ai-sdk-compaction.ts index f6967f2456..9eea7d0a37 100644 --- a/packages/runtime/src/ai-sdk-compaction.ts +++ b/packages/runtime/src/ai-sdk-compaction.ts @@ -845,6 +845,13 @@ export class AiSdkCompaction { queue: AsyncEventQueue, providerTools: readonly MakaTool[], onDiagnosticPatch: (patch: Partial) => void, + /** + * Fired once a compaction is durable and applied — checkpoint persisted and the + * projection takes over — before the replacement request goes out. The + * runtime records its apply-time boundary row here; like every system + * note it must never disturb the compaction, so the call is fail-open. + */ + onCompactionApplied: (() => void | Promise) | undefined, origin: ProviderRequestOrigin, memoryCompactionDecision?: () => AutomaticMemoryCompactionDecision, onMemoryCompaction?: (input: AutomaticMemoryCompactionDispatch) => void, @@ -932,6 +939,7 @@ export class AiSdkCompaction { activeToolsForStep, memoryCompactionDecision, onMemoryCompaction, + onCompactionApplied, abortSignal, }); if (outcome.decision === 'fail') { @@ -978,6 +986,8 @@ export class AiSdkCompaction { activeToolsForStep: readonly string[]; memoryCompactionDecision?: () => AutomaticMemoryCompactionDecision; onMemoryCompaction?: (input: AutomaticMemoryCompactionDispatch) => void; + /** See {@link buildMidTurnCapacityCompactProjection}'s parameter. */ + onCompactionApplied?: () => void | Promise; phase?: 'pre_turn' | 'mid_turn'; abortSignal?: AbortSignal; }): Promise { @@ -1193,6 +1203,14 @@ export class AiSdkCompaction { } } state.projectionCheckpoint = plan.checkpoint; + // The boundary is durable the moment the projection takes over: record it + // now, not at turn settlement, so usage readers ordering measurements + // against the compaction get the apply time. + try { + await input.onCompactionApplied?.(); + } catch { + // The boundary row explains the compaction; losing it must not undo one. + } return { decision: 'compacted', checkpoint: plan.checkpoint, @@ -1227,6 +1245,8 @@ export class AiSdkCompaction { origin: ProviderRequestOrigin; memoryCompactionDecision?: () => AutomaticMemoryCompactionDecision; onMemoryCompaction?: (input: AutomaticMemoryCompactionDispatch) => void; + /** See {@link buildMidTurnCapacityCompactProjection}'s parameter. */ + onCompactionApplied?: () => void | Promise; abortSignal?: AbortSignal; }): Promise<{ messages: ModelMessage[] } | undefined> { const state = input.midTurnState; @@ -1262,6 +1282,7 @@ export class AiSdkCompaction { activeToolsForStep: input.activeTools, memoryCompactionDecision: input.memoryCompactionDecision, onMemoryCompaction: input.onMemoryCompaction, + onCompactionApplied: input.onCompactionApplied, abortSignal: input.abortSignal, }); if (outcome.decision !== 'compacted') { @@ -1479,20 +1500,78 @@ function persistedRequestAnchor( modelId: string, connectionId: string | undefined, ): LastRequestAnchor | undefined { + // The newest anchored usage row is the candidate — but a send that compacted + // mid-turn and never completed another request settles its usage row AFTER + // the compaction notes while the anchor still describes the pre-compaction + // request (#5547). Position alone cannot order that pair, so the scan holds + // the candidate and lets a boundary row behind it supersede by event time. + let pending: { anchor: LastRequestAnchor; completedAt?: number } | undefined; for (let index = events.length - 1; index >= 0; index -= 1) { const event = events[index]; const anchor = event?.actions?.tokenUsage?.lastRequestAnchor; - if (!anchor) continue; - const route = invocations.find((candidate) => candidate.runId === event?.runId)?.opening.route; - if ( - route?.provenance !== 'runtime' || - route.backendKind === 'plugin-executor' || - route.modelId !== modelId || - route.llmConnectionId !== connectionId - ) { - return undefined; + if (anchor) { + // A second anchored row settles the candidate: under ordered writes every + // boundary behind it is strictly older than the candidate's completion. + if (pending) return pending.anchor; + const route = invocations.find((candidate) => candidate.runId === event?.runId)?.opening + .route; + if ( + route?.provenance !== 'runtime' || + route.backendKind === 'plugin-executor' || + route.modelId !== modelId || + route.llmConnectionId !== connectionId + ) { + return undefined; + } + pending = { anchor, completedAt: anchor.completedAt }; + continue; + } + const note = event?.content?.kind === 'system_note' ? event.content.note : undefined; + if (note === 'context_compaction_applied' || note === 'context_compacted') { + // A boundary behind the candidate is strictly older under ordered + // writes — unless the candidate's completion itself predates the compaction. + if (!pending) return undefined; + const boundaryAt = + note === 'context_compacted' + ? // The display row lands at settlement — after every request of + // its turn — so its own row time would supersede even an anchor + // that completed after the compaction. The compaction recorded its real + // boundary at apply time earlier in the same turn; only when no + // such row exists is the note's own time all there is. + (compactionAppliedAtBefore(events, index, event.turnId) ?? event.ts) + : event.ts; + return pending.completedAt !== undefined && + boundaryAt !== undefined && + boundaryAt >= pending.completedAt + ? undefined + : pending.anchor; + } + } + return pending?.anchor; +} + +/** + * The apply-time boundary a settlement `context_compacted` row describes: + * the same turn's `context_compaction_applied` note. Between the compaction's + * apply row and its display row sit only that turn's post-compaction output — + * the turn's own usage row lands after the note — so the first relevant + * row behind the note is the matching apply row when one was recorded. A + * usage row, another display note, a different turn's apply row, or the + * head of the log all mean the note's own write time is all there is. + */ +function compactionAppliedAtBefore( + events: readonly RuntimeEvent[], + noteIndex: number, + turnId: string | undefined, +): number | undefined { + for (let index = noteIndex - 1; index >= 0; index -= 1) { + const event = events[index]; + if (event?.actions?.tokenUsage !== undefined) return undefined; + const note = event?.content?.kind === 'system_note' ? event.content.note : undefined; + if (note === 'context_compaction_applied') { + return event?.turnId === turnId ? event.ts : undefined; } - return anchor; + if (note === 'context_compacted') return undefined; } return undefined; } diff --git a/packages/runtime/src/ai-sdk-turn.ts b/packages/runtime/src/ai-sdk-turn.ts index 459b40c63f..f8e2ccef2e 100644 --- a/packages/runtime/src/ai-sdk-turn.ts +++ b/packages/runtime/src/ai-sdk-turn.ts @@ -1002,6 +1002,10 @@ export class AiSdkTurn { // Output tokens of the same step: with the input they are the baseline the // next request is judged from (everything the model produced is re-sent). let lastStepOutputTokens: number | undefined; + // When that step's request finished: the settlement anchor carries it as + // `completedAt` so readers can order the measurement against a compaction's + // apply-time boundary (#5547) even where no provider tracker is wired. + let lastStepRequestCompletedAt: number | undefined; let streamStatus: LlmCallRecord['status'] = 'success'; let streamErrorClass: string | undefined; let runtimeSteps = 0; @@ -1046,6 +1050,12 @@ export class AiSdkTurn { contextCompactedNoteWritten = await this.recordSystemNote('context_compacted', turnId); } }; + // The apply-time boundary the usage reader orders measurements against: + // recorded when each compaction lands, so it exists mid-turn and a stop or + // stream error cannot erase it — unlike the settlement-time display note. + const recordCompactionApplied = async (): Promise => { + await this.recordSystemNote('context_compaction_applied', turnId); + }; const trace = new RunTrace({ sessionId: this.deps.backend.sessionId, turnId, @@ -1436,6 +1446,7 @@ export class AiSdkTurn { queue, capacityProviderTools, onMidTurnDiagnosticPatch, + recordCompactionApplied, this, this.automaticMemoryCompactionSupported() ? () => this.automaticMemoryCompactionDecision() @@ -1921,6 +1932,7 @@ export class AiSdkTurn { } lastStepInputTokens = stepUsage?.inputTokens; lastStepOutputTokens = stepUsage?.outputTokens; + lastStepRequestCompletedAt = this.deps.now(); // A `finishReason: length` is deliberately not a trigger. The // reply may have been cut because the provider ran out of // window room, or because the provider's own output cap is @@ -1992,6 +2004,7 @@ export class AiSdkTurn { activeTools: activeToolsForRequest, queue, onDiagnosticPatch: onMidTurnDiagnosticPatch, + onCompactionApplied: recordCompactionApplied, origin: this, ...(this.automaticMemoryCompactionSupported() ? { @@ -2359,6 +2372,14 @@ export class AiSdkTurn { // are what the next request re-sends. No usable input count, no // anchor: the next turn then has no proactive fold until its first // accepted request. + const lastCompletedMainRequestAt = + providerRequestTracker?.latestCompletedMainRequestAt ?? lastStepRequestCompletedAt; + // This row lands after the compaction notes even when the + // anchored request completed BEFORE the compaction — the post-compaction + // retry never finished (#5547). The anchor is persisted whole as + // the durable fact it is; readers order it against the compaction's + // apply-time boundary via `completedAt` rather than trusting + // ledger position. const anchorInputTokens = finitePositive(lastStepInputTokens); const anchorOutputTokens = lastStepOutputTokens !== undefined && Number.isFinite(lastStepOutputTokens) @@ -2396,6 +2417,9 @@ export class AiSdkTurn { ? { lastRequestAnchor: { inputTokens: anchorInputTokens, + ...(lastCompletedMainRequestAt !== undefined + ? { completedAt: lastCompletedMainRequestAt } + : {}), ...(anchorOutputTokens !== undefined ? { outputTokens: anchorOutputTokens } : {}), diff --git a/packages/runtime/src/provider-request-telemetry.ts b/packages/runtime/src/provider-request-telemetry.ts index 05272b153f..3913dd6592 100644 --- a/packages/runtime/src/provider-request-telemetry.ts +++ b/packages/runtime/src/provider-request-telemetry.ts @@ -428,6 +428,7 @@ function modelCallUsageFields( export class ProviderRequestTracker { private step = 0; + private lastCompletedMainRequestAt: number | undefined; private readonly attemptsByStep = new Map(); /** * One logical call per step. Retries of the same step are further attempts of @@ -442,6 +443,10 @@ export class ProviderRequestTracker { return this.input.traceId; } + get latestCompletedMainRequestAt(): number | undefined { + return this.lastCompletedMainRequestAt; + } + setStep(step: number, requestCompositionId?: string): void { this.step = step; if (requestCompositionId !== undefined) { @@ -617,6 +622,9 @@ export class ProviderRequestTracker { input.abortSignal?.removeEventListener('abort', abortListener); } const completedAt = this.input.now(); + if (status === 'completed' && this.input.accounting?.callKind === 'main') { + this.lastCompletedMainRequestAt = completedAt; + } const usage = strictProviderRequestUsage(finish?.usage); const contextWindow = positiveInteger(this.input.contextWindow); const diagnostic = diff --git a/packages/runtime/src/runtime-kernel.ts b/packages/runtime/src/runtime-kernel.ts index 4bada91831..3107b9252d 100644 --- a/packages/runtime/src/runtime-kernel.ts +++ b/packages/runtime/src/runtime-kernel.ts @@ -1306,6 +1306,9 @@ export class RuntimeKernel implements RuntimeKernelLike { // next user send passively replays this standalone checkpoint, which // `shouldAppendContextCompactedNote` suppresses, so there is no // duplicate. + // The applied row is the apply-time boundary the usage reader orders + // measurements against; the note below stays the display row. + await run.recordSystemNote('context_compaction_applied').catch(() => {}); await run.recordSystemNote('context_compacted').catch(() => {}); notedTerminal = true; } diff --git a/packages/ui/src/__tests__/composer-context-usage.test.tsx b/packages/ui/src/__tests__/composer-context-usage.test.tsx index 9f6c229f40..5791d88396 100644 --- a/packages/ui/src/__tests__/composer-context-usage.test.tsx +++ b/packages/ui/src/__tests__/composer-context-usage.test.tsx @@ -23,6 +23,7 @@ import { act } from 'react'; import { createRoot } from 'react-dom/client'; import { parseHTML } from 'linkedom'; import { Composer } from '../composer.js'; +import type { ContextUsageReading } from '../context-usage-reading.js'; import { LocaleProvider } from '../locale-context.js'; test('the context usage action opens its host trace surface', async () => { @@ -49,7 +50,7 @@ test('the context usage action opens its host trace surface', async () => { await act(() => root.render( { opened = true; } }} + contextUsage={{ reading: { kind: 'unavailable' }, onOpen: () => { opened = true; } }} onSend={() => undefined} onStop={() => undefined} /> @@ -99,11 +100,11 @@ test('the context usage share resolves declared, then metered, then metadata win const render = async ( contextUsage: { - usageTokens?: number; + reading: ContextUsageReading; declaredContextWindow?: number; - meteredContextWindow?: number; metadataContextWindow?: number; }, + expectedTooltip?: string, ) => { await act(() => root.render( @@ -118,6 +119,13 @@ test('the context usage share resolves declared, then metered, then metadata win 'button[aria-label="Open usage trace"]', ); assert.ok(action); + if (expectedTooltip !== undefined) { + const describedBy = action.getAttribute('aria-describedby'); + assert.ok(describedBy, 'context usage must reference its tooltip'); + const tooltip = document.getElementById(describedBy); + assert.ok(tooltip, 'context usage tooltip must exist'); + assert.equal(tooltip.textContent?.trim(), expectedTooltip); + } return action.textContent?.trim(); }; @@ -125,9 +133,8 @@ test('the context usage share resolves declared, then metered, then metadata win // The user's declaration wins over every reported window. assert.equal( await render({ - usageTokens: 40_000, + reading: { kind: 'measured', tokens: 40_000, meteredWindow: 80_000 }, declaredContextWindow: 100_000, - meteredContextWindow: 80_000, metadataContextWindow: 64_000, }), '40%', @@ -135,13 +142,27 @@ test('the context usage share resolves declared, then metered, then metadata win // The metered window was frozen against the same request as the tokens, // so it outranks the catalog's metadata window. assert.equal( - await render({ usageTokens: 40_000, meteredContextWindow: 80_000, metadataContextWindow: 64_000 }), + await render({ reading: { kind: 'measured', tokens: 40_000, meteredWindow: 80_000 }, metadataContextWindow: 64_000 }), '50%', ); // Metadata is the fallback… - assert.equal(await render({ usageTokens: 32_000, metadataContextWindow: 64_000 }), '50%'); + assert.equal(await render({ reading: { kind: 'measured', tokens: 32_000 }, metadataContextWindow: 64_000 }), '50%'); // …and with no window at all the usage stands alone, no invented share. - assert.equal(await render({ usageTokens: 40_000 }), 'Usage'); + assert.equal(await render({ reading: { kind: 'measured', tokens: 40_000 } }), 'Usage'); + // A superseded reading keeps the usage entry label even when a window is known. + assert.equal(await render( + { reading: { kind: 'stale', reason: 'compaction' }, declaredContextWindow: 100_000 }, + 'Context has been compacted. Usage will update when the next request completes.', + ), 'Usage'); + // A later successful measurement restores the share in the same mounted control. + assert.equal(await render( + { reading: { kind: 'measured', tokens: 10_000, meteredWindow: 100_000 } }, + 'Context window: 10% used (10K / 100K tokens).', + ), '10%'); + assert.equal(await render( + { reading: { kind: 'unavailable' }, declaredContextWindow: 100_000 }, + 'No usage data is available for this request.', + ), 'Usage'); } finally { await act(() => root.unmount()); Object.assign(globalThis, original); diff --git a/packages/ui/src/__tests__/materialize.test.ts b/packages/ui/src/__tests__/materialize.test.ts index b9c8c62d78..115cce2a09 100644 --- a/packages/ui/src/__tests__/materialize.test.ts +++ b/packages/ui/src/__tests__/materialize.test.ts @@ -257,6 +257,31 @@ describe("materializeTurns message metadata", () => { assert.deepEqual(turns.flatMap((turn) => turn.notes), []); }); + test("hides the apply-time compaction boundary note that exists for usage ordering", () => { + // The applied row is the selector's compaction-time anchor; the settlement + // context_compacted row stays the one users see. + const turns = materializeTurns([ + { + type: "system_note", + id: "applied", + turnId: "t1", + ts: 1, + kind: "context_compaction_applied", + }, + { + type: "system_note", + id: "compacted", + turnId: "t1", + ts: 2, + kind: "context_compacted", + }, + ], "en"); + assert.deepEqual( + turns.flatMap((turn) => turn.notes.map((note) => note.id)), + ["compacted"], + ); + }); + test("localizes visible system notes", () => { const messages: StoredMessage[] = [ { diff --git a/packages/ui/src/composer.tsx b/packages/ui/src/composer.tsx index d2c9b71f39..a467e330d6 100644 --- a/packages/ui/src/composer.tsx +++ b/packages/ui/src/composer.tsx @@ -33,6 +33,7 @@ import { type KeyboardEvent, type ReactNode, } from 'react'; +import type { ContextUsageReading } from './context-usage-reading.js'; import type { LucideIcon } from './icons.js'; import { useMountedRef } from './use-mounted-ref.js'; import { isAppleShortcutPlatform } from './utils.js'; @@ -476,14 +477,8 @@ export const Composer = forwardRef< noModelHint?: string; /** Read-only usage indicator for the active model's latest request. */ contextUsage?: { - usageTokens?: number; + reading: ContextUsageReading; declaredContextWindow?: number; - /** - * The window the usage number was metered against, frozen at call time. - * When present it outranks the metadata window, so a live reading keeps - * its numerator and denominator from the same request. - */ - meteredContextWindow?: number; metadataContextWindow?: number; /** Open the Host-owned trace surface for this readout. */ onOpen(): void; @@ -2663,13 +2658,15 @@ export const Composer = forwardRef< }); function ContextUsageAction(props: { - usageTokens?: number; + reading: ContextUsageReading; declaredContextWindow?: number; - meteredContextWindow?: number; metadataContextWindow?: number; onOpen(): void; }) { const copy = getConversationCopy(useUiLocale()).messages; + const { reading } = props; + const usageTokens = reading.kind === 'measured' ? reading.tokens : undefined; + const meteredWindow = reading.kind === 'measured' ? reading.meteredWindow : undefined; // A window from any source is enough to show a share, and the order is a // claim about which window the number was earned against: the user's // declaration first — it is the user's intent, and the only one that arms @@ -2678,17 +2675,18 @@ function ContextUsageAction(props: { // same request, and only then the model's reported metadata. With no window // at all the usage stands on its own. const window = - props.declaredContextWindow ?? props.meteredContextWindow ?? props.metadataContextWindow; - const label = - props.usageTokens !== undefined && window !== undefined && window > 0 - ? `${Math.round((props.usageTokens / window) * 100)}%` + props.declaredContextWindow ?? meteredWindow ?? props.metadataContextWindow; + // Without a current measurement, keep the usage entry label. + const label = usageTokens !== undefined && window !== undefined && window > 0 + ? `${Math.round((usageTokens / window) * 100)}%` : copy.systemNotes.contextUsageLabel; - const tooltip = - props.usageTokens === undefined + const tooltip = reading.kind === 'stale' + ? copy.systemNotes.contextUsageCompacted + : usageTokens === undefined ? copy.systemNotes.contextUsageUnavailable : window !== undefined && window > 0 - ? copy.systemNotes.contextUsageShare(props.usageTokens, window) - : copy.systemNotes.contextUsageNoWindow(props.usageTokens); + ? copy.systemNotes.contextUsageShare(usageTokens, window) + : copy.systemNotes.contextUsageNoWindow(usageTokens); return ( string; contextUsageNoWindow: (used: number) => string; contextUsageUnavailable: string; + contextUsageCompacted: string; contextUsageOpen: string; stepLimit: string; }; @@ -541,6 +542,7 @@ const CONVERSATION_COPY = { contextUsageNoWindow: (used) => `已用 ${formatCompactTokenCount(used)} token;上下文窗口上限未知`, contextUsageUnavailable: '暂无用量数据', + contextUsageCompacted: '上下文已压缩,用量将在下一次请求完成后更新。', contextUsageOpen: '打开用量追踪', stepLimit: '已达到本轮工具步骤上限,任务可能尚未完成。发送“继续”即可接着处理。', }, @@ -666,6 +668,7 @@ const CONVERSATION_COPY = { contextUsageNoWindow: (used) => `已用 ${formatCompactTokenCount(used)} token;上下文視窗上限未知`, contextUsageUnavailable: '暫無用量資料', + contextUsageCompacted: '上下文已壓縮,用量將在下一次請求完成後更新。', contextUsageOpen: '開啟用量追蹤', stepLimit: '已達到本輪工具步驟上限,任務可能尚未完成。傳送“繼續”即可接著處理。', }, @@ -788,6 +791,8 @@ const CONVERSATION_COPY = { contextUsageNoWindow: (used) => `This request used ${formatCompactTokenCount(used)} tokens; no context limit is available for this model.`, contextUsageUnavailable: 'No usage data is available for this request.', + contextUsageCompacted: + 'Context has been compacted. Usage will update when the next request completes.', contextUsageOpen: 'Open usage trace', stepLimit: 'Reached the configured step limit. The task may be incomplete. Send “continue” to resume.', }, diff --git a/packages/ui/src/index.ts b/packages/ui/src/index.ts index 98063c96ba..f50230f305 100644 --- a/packages/ui/src/index.ts +++ b/packages/ui/src/index.ts @@ -39,6 +39,7 @@ export type { export type { SessionMoveTarget } from './session-rail-context.js'; export * from './session-status-presentation.js'; export * from './composer-helpers.js'; +export type { ContextUsageReading } from './context-usage-reading.js'; export * from './conversation-copy.js'; export * from './shared-ui-copy.js'; export * from './skills-copy.js';