diff --git a/apps/desktop/src/main/__tests__/session-inspector-usage-stats.test.ts b/apps/desktop/src/main/__tests__/session-inspector-usage-stats.test.ts index 14516d4bb1..17a3a27e37 100644 --- a/apps/desktop/src/main/__tests__/session-inspector-usage-stats.test.ts +++ b/apps/desktop/src/main/__tests__/session-inspector-usage-stats.test.ts @@ -27,6 +27,11 @@ import { usageRingArcs, } from '../../renderer/features/workbar/testing.js'; import type { UsageSummaryV2 } from '@maka/core/usage-stats/types'; +import type { DesktopSessionUsageSummary } from '../../preload/bridge-contract.js'; +import { + loadSessionUsageSummaryVia, + type UsageSummaryInvoke, +} from '../../preload/usage-summary.js'; function usageSummary(overrides: Partial = {}): UsageSummaryV2 { return { @@ -50,6 +55,22 @@ function usageSummary(overrides: Partial = {}): UsageSummaryV2 { }; } +function fullCoverageProvenance(): DesktopSessionUsageSummary['provenance'] { + return { + coverage: { + attempts: 3, + pricedAttempts: 3, + unpricedAttempts: 0, + usageReportedAttempts: 3, + usagePartialAttempts: 0, + usageMissingAttempts: 0, + }, + legacyRecords: 0, + unreadableRecords: 0, + pendingRepairs: 0, + }; +} + test('splits the session metered tokens the way a bill reads', () => { const { tokenUsage } = deriveInspectorOverviewModel( undefined, @@ -353,3 +374,132 @@ test('carries rounded inspector durations into the next unit', () => { assert.equal(formatDuration(3_599_600), '60m0s'); assert.equal(formatDuration(125_000), '2m5s'); }); + +test('reads the cache rate from the main loop alone when the main summary is available', () => { + // A 10k-token uncached suggestion call drops a 95%-cached main loop to a + // blended 86%: auxiliary prompts keep their own prefix, so the rate must + // read the main summary, not the blend (#5691). + const blended = usageSummary({ + totalTokens: { + input: 1_100_000, + output: 60_300, + cacheMiss: 150_000, + cacheRead: 950_000, + cacheWrite: 0, + reasoning: 0, + total: 1_160_300, + }, + }); + const main = usageSummary({ + totalTokens: { + input: 1_000_000, + output: 60_000, + cacheMiss: 50_000, + cacheRead: 950_000, + cacheWrite: 0, + reasoning: 0, + total: 1_060_000, + }, + }); + + const blendedRate = deriveInspectorOverviewModel(undefined, blended).cacheHitRate; + const refined = deriveInspectorOverviewModel(undefined, { + ...blended, + mainSummary: main, + }).cacheHitRate; + + assert.ok(blendedRate); + assert.equal(Math.round(blendedRate * 100), 86); + assert.ok(refined); + assert.equal(Math.round(refined * 100), 95); +}); + +test('shows no cache rate when the main-only read failed rather than blending', () => { + // A failed narrow query leaves the blended totals in place. Reporting the + // blended 86% as the main loop's rate is the same misreading the main + // summary exists to prevent (#5691 review): hide the rate instead. + const { cacheHitRate } = deriveInspectorOverviewModel(undefined, { + ...usageSummary({ + totalTokens: { + input: 1_100_000, + output: 60_300, + cacheMiss: 150_000, + cacheRead: 950_000, + cacheWrite: 0, + reasoning: 0, + total: 1_160_300, + }, + }), + mainSummaryUnavailable: true, + }); + + assert.equal(cacheHitRate, undefined); +}); + +test('marks the overview when the injected preload IPC read of the main loop fails', async () => { + // The model-level test above pins the hiding rule; this one pins the + // preload wiring that feeds it. A failed main-only IPC call — the injected + // preload failure path — must mark the blended summary rather than drop + // the overview or present the blend as the main loop's (#5691 review). + const blended: DesktopSessionUsageSummary = { + ...usageSummary(), + provenance: fullCoverageProvenance(), + }; + const calls: Array> = []; + const invoke: UsageSummaryInvoke = async (_channel, _scope, args) => { + calls.push(args); + return (args as { callKinds?: readonly string[] }).callKinds + ? { ok: false, error: { code: 'persistence_failed', message: 'usage read failed' } } + : { ok: true, data: blended }; + }; + + const outcome = await loadSessionUsageSummaryVia(invoke, { + scope: { profileId: 'profile-1' }, + sessionId: 'session-1', + }); + + assert.equal(outcome.ok, true); + if (!outcome.ok) return; + assert.equal(outcome.data.mainSummary, undefined); + assert.equal(outcome.data.mainSummaryUnavailable, true); + + // A crashed secondary invoke (rejected promise, not an ok:false result) + // must degrade the same way instead of failing the overview (#5691 review). + const crashingInvoke: UsageSummaryInvoke = async (_channel, _scope, args) => { + if ((args as { callKinds?: readonly string[] }).callKinds) { + throw new Error('IPC transport died'); + } + return { ok: true, data: usageSummary() } as Awaited>; + }; + const degraded = await loadSessionUsageSummaryVia(crashingInvoke, { + scope: { profileId: 'profile-1' }, + sessionId: 'session-1', + }); + assert.equal(degraded.ok, true); + if (!degraded.ok) return; + assert.equal(degraded.data.mainSummaryUnavailable, true); + // The second read asked for exactly the agent loop's own calls. + assert.deepEqual( + calls.map((args) => (args as { callKinds?: readonly string[] }).callKinds ?? null), + [null, ['main']], + ); +}); + +test('attaches the main summary when both injected preload reads succeed', async () => { + const invoke: UsageSummaryInvoke = async (_channel, _scope, args) => ({ + ok: true, + data: (args as { callKinds?: readonly string[] }).callKinds + ? { ...usageSummary(), provenance: fullCoverageProvenance(), cacheHitRequests: 2 } + : { ...usageSummary({ cacheHitRequests: 3 }), provenance: fullCoverageProvenance() }, + }); + + const outcome = await loadSessionUsageSummaryVia(invoke, { + scope: { profileId: 'profile-1' }, + sessionId: 'session-1', + }); + + assert.equal(outcome.ok, true); + if (!outcome.ok) return; + assert.equal(outcome.data.mainSummaryUnavailable, undefined); + assert.equal(outcome.data.mainSummary?.cacheHitRequests, 2); +}); diff --git a/apps/desktop/src/preload/bridge-contract.d.ts b/apps/desktop/src/preload/bridge-contract.d.ts index 6cca3cf51a..7f0add2e0d 100644 --- a/apps/desktop/src/preload/bridge-contract.d.ts +++ b/apps/desktop/src/preload/bridge-contract.d.ts @@ -806,6 +806,14 @@ export interface DesktopSessionTracePage { export interface DesktopSessionUsageSummary extends UsageSummaryV2 { readonly provenance: UsageProvenance; + /** + * The same Session scoped to the agent loop's own calls (`callKinds: + * ['main']`), when the narrower read succeeded. The overview's cache rate + * reads this: auxiliary prompts have their own cache prefix (#5691). + */ + readonly mainSummary?: DesktopSessionUsageSummary; + /** The narrower read failed; the blended rate must not stand in for it. */ + readonly mainSummaryUnavailable?: boolean; } export interface MakaBridge { diff --git a/apps/desktop/src/preload/preload.ts b/apps/desktop/src/preload/preload.ts index 9eee67caeb..027259b8d1 100644 --- a/apps/desktop/src/preload/preload.ts +++ b/apps/desktop/src/preload/preload.ts @@ -92,6 +92,7 @@ import type { AppIconSelectResult, } from './bridge-contract.js'; import type { ExternalSessionImportIpcResult } from './external-session-import-result.js'; +import { loadSessionUsageSummaryVia } from './usage-summary.js'; import type { RuntimeHostObservationIpcResult } from '../shared/runtime-host-observation-ipc.js'; import { projectDesktopExternalSessionCatalogItem, @@ -1359,12 +1360,7 @@ async function loadSessionTracePage( async function loadSessionUsageSummary( sessionId: string, ): Promise> { - const session = await runtimeHostSessionRef(sessionId); - return invokeWhenReady( - 'usage:summary', - session.scope, - { range: 'all', sessionId: session.sessionId }, - ) as Promise>; + return loadSessionUsageSummaryVia(invokeWhenReady, await runtimeHostSessionRef(sessionId)); } async function updateDailyReviewConfig( diff --git a/apps/desktop/src/preload/usage-summary.ts b/apps/desktop/src/preload/usage-summary.ts new file mode 100644 index 0000000000..d62c2cbb06 --- /dev/null +++ b/apps/desktop/src/preload/usage-summary.ts @@ -0,0 +1,58 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +import type { Result } from '@maka/core/result'; +import type { DesktopSessionUsageSummary } from './bridge-contract.js'; + +/** The narrow invoke surface the usage overview needs; injectable for tests. */ +export type UsageSummaryInvoke = ( + channel: 'usage:summary', + scope: unknown, + args: Record, +) => Promise>; + +/** + * The overview reads the Session's whole metered spend and, separately, the + * agent loop's own calls: auxiliary prompts do not share the main loop's + * cached prefix, so a blended rate under-reports it (#5691). + * + * The main-only read refines the cache rate; losing it must not lose the + * overview — but it must also not pass the blended rate off as the main + * loop's, so the failure is marked and the rate hides itself (#5691). + */ +export async function loadSessionUsageSummaryVia( + invoke: UsageSummaryInvoke, + session: { readonly scope: unknown; readonly sessionId: string }, +): Promise> { + const summaryQuery = { range: 'all' as const, sessionId: session.sessionId }; + // A rejected secondary invoke (transport/IPC crash rather than an + // operation-level failure) must not take the whole overview down with it — + // it marks the summary unavailable like any other main-read failure (#5691 + // review). + const [summary, main] = (await Promise.all([ + invoke('usage:summary', session.scope, summaryQuery), + invoke('usage:summary', session.scope, { ...summaryQuery, callKinds: ['main'] }).catch( + () => ({ ok: false, error: { code: 'persistence_failed', message: 'usage read failed' } }), + ), + ])) as [Result, Result]; + if (!summary.ok) return summary; + return main.ok + ? { ...summary, data: { ...summary.data, mainSummary: main.data } } + : { ...summary, data: { ...summary.data, mainSummaryUnavailable: true } }; +} diff --git a/apps/desktop/src/renderer/application/contracts/session-inspector/session-inspector-overview-model.ts b/apps/desktop/src/renderer/application/contracts/session-inspector/session-inspector-overview-model.ts index b4af6d30cd..3598c9d90a 100644 --- a/apps/desktop/src/renderer/application/contracts/session-inspector/session-inspector-overview-model.ts +++ b/apps/desktop/src/renderer/application/contracts/session-inspector/session-inspector-overview-model.ts @@ -24,7 +24,13 @@ import type { import type { UsageSummaryV2 } from '@maka/core/usage-stats/types'; import type { UsageProvenance } from '@maka/core/usage-ledger-merge'; -type SessionUsageSummary = UsageSummaryV2 & { readonly provenance?: UsageProvenance }; +type SessionUsageSummary = UsageSummaryV2 & { + readonly provenance?: UsageProvenance; + /** The agent loop's own calls, when the narrower read succeeded (#5691). */ + readonly mainSummary?: SessionUsageSummary; + /** The narrower read failed; the blended rate must not stand in for it. */ + readonly mainSummaryUnavailable?: boolean; +}; /** * Overview view model for the Inspector panel's summary sections. @@ -243,15 +249,24 @@ export function deriveInspectorOverviewModel( } function usageCacheHitRate(usage: SessionUsageSummary | undefined): number | undefined { - if (!usage || usage.totalTokens.input === 0) return undefined; + // Auxiliary calls keep their own prompt prefix, so blending them into the + // rate reports the main loop's caching as worse than it is (#5691). The + // main-only summary is the rate's input. When that narrower read failed, + // show no rate at all: the blended number is about a different set of calls + // and presenting it as the main loop's is the same lie in weaker type. The + // fallback to the blended summary only serves summaries that never asked + // the narrower question. + if (usage?.mainSummaryUnavailable && !usage.mainSummary) return undefined; + const main = usage?.mainSummary ?? usage; + if (!main || main.totalTokens.input === 0) return undefined; if ( - usage.provenance && - (usage.provenance.coverage.usagePartialAttempts > 0 || - usage.provenance.coverage.usageMissingAttempts > 0) + main.provenance && + (main.provenance.coverage.usagePartialAttempts > 0 || + main.provenance.coverage.usageMissingAttempts > 0) ) { return undefined; } - return usage.totalTokens.cacheRead / usage.totalTokens.input; + return main.totalTokens.cacheRead / main.totalTokens.input; } /** diff --git a/packages/core/src/usage-stats/types.ts b/packages/core/src/usage-stats/types.ts index bd803b138b..1a56088c94 100644 --- a/packages/core/src/usage-stats/types.ts +++ b/packages/core/src/usage-stats/types.ts @@ -48,6 +48,13 @@ export interface UsageQuery { modelId?: string; toolName?: string; status?: 'success' | 'error' | 'aborted' | 'all'; + /** + * Keep only calls of these kinds. Unset means every kind, which is what a + * Session's headline cost has always reported; `['main']` is how a consumer + * reads the agent loop's own spend and cache behavior without the auxiliary + * calls recorded alongside it (#5691). + */ + callKinds?: readonly ModelCallKind[]; } export interface UsageSummaryV2 { diff --git a/packages/runtime-host/protocol-compatible-changes/mechanical-candidate-sweep.json b/packages/runtime-host/protocol-compatible-changes/mechanical-candidate-sweep.json index a4e428f9d6..173d280e06 100644 --- a/packages/runtime-host/protocol-compatible-changes/mechanical-candidate-sweep.json +++ b/packages/runtime-host/protocol-compatible-changes/mechanical-candidate-sweep.json @@ -1,5 +1,5 @@ { - "epoch": 200, + "epoch": 208, "files": [ "packages/runtime-host/src/protocol/operations.ts", "packages/runtime-host/src/protocol/turn.ts" diff --git a/packages/runtime-host/protocol-compatible-changes/turn-snapshot-optional-fields.json b/packages/runtime-host/protocol-compatible-changes/turn-snapshot-optional-fields.json index be183f2c6c..69f49fa605 100644 --- a/packages/runtime-host/protocol-compatible-changes/turn-snapshot-optional-fields.json +++ b/packages/runtime-host/protocol-compatible-changes/turn-snapshot-optional-fields.json @@ -1,5 +1,5 @@ { - "epoch": 200, + "epoch": 208, "files": ["packages/runtime-host/src/protocol/turn.ts"], "reason": "Reorganizes the local construction of already-supported optional Turn snapshot fields; it changes neither the wire keys nor accepted values." } diff --git a/packages/runtime-host/src/protocol/index.ts b/packages/runtime-host/src/protocol/index.ts index 65cfe86d24..060ead6c36 100644 --- a/packages/runtime-host/src/protocol/index.ts +++ b/packages/runtime-host/src/protocol/index.ts @@ -104,7 +104,9 @@ export const RUNTIME_HOST_REGISTRATION_SCHEMA_VERSION = 1 as const; export const RUNTIME_HOST_PROTOCOL_VERSION = 0 as const; // Increment when the same protocol version no longer guarantees safe Client-Host // interoperability. Mismatches are rejected before domain commands are admitted. -export const RUNTIME_HOST_COMPATIBILITY_EPOCH = 205 as const; +export const RUNTIME_HOST_COMPATIBILITY_EPOCH = 208 as const; +// 208: Usage queries filter by model call kind. Epoch-207 peers reject the +// filter or ignore it. // 205: Default-model selection can enable atomically; discovery can preserve the model selection. // 204: Executor catalogs and Session configuration carry opaque mode IDs; // catalog queries may request a provider refresh. Older peers reject these fields. diff --git a/packages/runtime-host/src/protocol/usage-pricing.ts b/packages/runtime-host/src/protocol/usage-pricing.ts index 20be8381a2..d9dfd135d9 100644 --- a/packages/runtime-host/src/protocol/usage-pricing.ts +++ b/packages/runtime-host/src/protocol/usage-pricing.ts @@ -67,6 +67,7 @@ const LLM_USAGE_QUERY_FIELDS = new Set([ 'providerId', 'modelId', 'status', + 'callKinds', ]); const TOOL_USAGE_QUERY_FIELDS = new Set(['range', 'toolName', 'status']); const USAGE_BUCKET_FIELDS = new Set([ @@ -709,9 +710,20 @@ function decodeLlmUsageQuery(value: unknown): LlmUsageQuery { ...optionalQueryText(query, 'providerId'), ...optionalQueryText(query, 'modelId'), ...(query.status === undefined ? {} : { status: decodeUsageStatus(query.status) }), + ...(query.callKinds === undefined ? {} : { callKinds: decodeCallKinds(query.callKinds) }), }; } +function decodeCallKinds(value: unknown): readonly ModelCallKind[] { + if (!Array.isArray(value)) throw invalidProtocolFrame('Invalid usage query callKinds'); + return value.map((entry) => { + if (typeof entry !== 'string' || !MODEL_CALL_KINDS.includes(entry as ModelCallKind)) { + throw invalidProtocolFrame('Invalid usage query callKind'); + } + return entry as ModelCallKind; + }); +} + function decodeToolUsageQuery(value: unknown): ToolUsageQuery { const query = requireRecord(value, 'tool usage query'); assertAllowedKeys(query, TOOL_USAGE_QUERY_FIELDS, 'tool usage query'); diff --git a/packages/storage/src/__tests__/model-call-usage-query.test.ts b/packages/storage/src/__tests__/model-call-usage-query.test.ts index 421650c210..195f41576c 100644 --- a/packages/storage/src/__tests__/model-call-usage-query.test.ts +++ b/packages/storage/src/__tests__/model-call-usage-query.test.ts @@ -182,6 +182,29 @@ describe('Usage answers over the canonical ledger', () => { ); }); + test('filters by call kind, so a Session can read its main loop alone', async () => { + await withProjectedAttempts( + [ + attempt({ attemptId: 'main-1', callKind: 'main', inputTokens: 1000 }), + attempt({ attemptId: 'main-2', callKind: 'main', inputTokens: 1000 }), + attempt({ attemptId: 'title', callKind: 'session_title', inputTokens: 1000 }), + ], + async (ledger) => { + const requests = (query: Parameters[0]) => + ledger.summary(query, NOW).projection.totalRequests; + assert.equal(requests({ range: 'all' }), 3); + assert.equal(requests({ range: 'all', callKinds: ['main'] }), 2); + assert.equal(requests({ range: 'all', callKinds: ['session_title'] }), 1); + assert.equal(requests({ range: 'all', callKinds: ['main', 'session_title'] }), 3); + assert.equal( + requests({ range: 'all', callKinds: [] }), + 0, + 'an empty allowlist addresses nothing rather than everything', + ); + }, + ); + }); + test('interrupted counts as aborted, not as an error', async () => { // Collapsing a cut-short call into `error` would inflate the error rate // with user cancellations. diff --git a/packages/storage/src/__tests__/usage-stores.test.ts b/packages/storage/src/__tests__/usage-stores.test.ts index 3e7e53178f..023f5f4cd0 100644 --- a/packages/storage/src/__tests__/usage-stores.test.ts +++ b/packages/storage/src/__tests__/usage-stores.test.ts @@ -485,6 +485,34 @@ describe('InteractiveUsageStores', () => { }); }); + test('legacy summary filters LLM rows by call kind', async () => { + await withInteractiveRoot(async ({ capability }) => { + const owner = await tryAcquireInteractiveRootOwner(capability); + assert(owner); + const stores = await openInteractiveUsageStoresForWrite(owner.lease); + await stores.telemetry.recordLlmCall( + llmRecord({ id: 'main-call', callKind: 'main', inputTokens: 100 }), + ); + await stores.telemetry.recordLlmCall( + llmRecord({ id: 'title-call', callKind: 'session_title', inputTokens: 100 }), + ); + await stores.telemetry.recordLlmCall(llmRecord({ id: 'untagged-call', inputTokens: 100 })); + + const mainOnly = await stores.telemetry.summary({ + range: 'all', + callKinds: ['main'], + }); + assert.equal(mainOnly.totalRequests, 1); + assert.equal(mainOnly.totalTokens.input, 100); + + const everything = await stores.telemetry.summary({ range: 'all' }); + assert.equal(everything.totalRequests, 3); + + await stores.close(); + await owner.close(); + }); + }); + test('legacy summary sums recorded call time over the same rows as its tokens', async () => { await withInteractiveRoot(async ({ capability }) => { const owner = await tryAcquireInteractiveRootOwner(capability); diff --git a/packages/storage/src/model-call-usage-sql.ts b/packages/storage/src/model-call-usage-sql.ts index f7c26a86a0..e44ab50d69 100644 --- a/packages/storage/src/model-call-usage-sql.ts +++ b/packages/storage/src/model-call-usage-sql.ts @@ -95,6 +95,20 @@ export function countableFilter( equals('provider_id', query.providerId); equals('model_id', query.modelId); equals('connection_slug', query.connectionSlug); + if (query.callKinds !== undefined) { + if (query.callKinds.length === 0) { + // An empty allowlist addresses no rows at all; SQL has no `IN ()`, so + // the honest translation is a clause that is never true. + clauses.push('0'); + } else { + // Rows written before call_kind was recorded (call_kind IS NULL) are + // legacy and cannot be classified: a callKinds filter excludes them on + // purpose, so pre-callKind Sessions report only their newer calls + // (#5691 review). + clauses.push(`call_kind IN (${query.callKinds.map(() => '?').join(', ')})`); + parameters.push(...query.callKinds); + } + } if (query.status !== undefined && query.status !== 'all') { // `interrupted` joins `aborted`: both mean the call stopped short without // the provider reporting a failure. diff --git a/packages/storage/src/sqlite-usage-store.ts b/packages/storage/src/sqlite-usage-store.ts index 2cb66c03e3..c8c7d33e42 100644 --- a/packages/storage/src/sqlite-usage-store.ts +++ b/packages/storage/src/sqlite-usage-store.ts @@ -306,6 +306,12 @@ class SqliteTelemetryRepo implements TelemetryRepo { if (query.providerId && row.providerId !== query.providerId) return false; if (query.modelId && row.modelId !== query.modelId) return false; if (query.status && query.status !== 'all' && row.status !== query.status) return false; + if ( + query.callKinds !== undefined && + (row.callKind === undefined || !query.callKinds.includes(row.callKind)) + ) { + return false; + } return true; }); }