From 28ca7fe24b91442c29f0037398ca8791591d6d9b Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Thu, 17 Sep 2026 21:32:05 +0900 Subject: [PATCH 01/17] fix(runtime): preserve client capability tool outcomes Carry explicit CU and Client Capability success, error, and aborted outcomes through transport, durable events, continuity, and UI projections. Keep legacy tool result semantics and old records readable. Generated-by: OpenAI Codex --- .../src/main/__tests__/browser-tools.test.ts | 1 + .../mcp-form-host-integration.test.ts | 4 +- .../main/__tests__/mcp-runtime-e2e.test.ts | 2 + .../runtime-host-desktop-candidate.test.ts | 4 +- .../runtime-host-native-capabilities.test.ts | 26 +++++++-- .../main/computer-use-real-model-policy.ts | 5 ++ .../main/runtime-host-native-capabilities.ts | 16 ++++-- .../acp-session-event-mapper.test.ts | 26 +++++++++ .../mcp-form-host-integration.test.ts | 10 +++- ...e-host-capability-provider-command.test.ts | 5 +- .../tui-mcp-local-integration.test.ts | 2 + .../tui-mcp-remote-integration.test.ts | 5 +- packages/cli/src/acp/tool-event-mapper.ts | 7 ++- packages/cli/src/mcp-capability-provider.ts | 1 + packages/cli/src/pi-transcript.ts | 6 +- .../tool-result-record-schema.test.ts | 12 ++++ packages/core/src/events.ts | 2 + packages/core/src/mcp.ts | 2 + packages/core/src/runtime-event.ts | 7 ++- packages/core/src/session.ts | 6 ++ packages/core/src/tool-result-status.ts | 14 +++++ .../__tests__/authenticated-websocket.test.ts | 5 +- ...t-capability-admission-integration.test.ts | 10 +++- .../client-capability-channel.test.ts | 36 ++++++++---- .../client-capability-coordinator.test.ts | 12 ++-- ...ient-capability-interaction-broker.test.ts | 10 +++- ...lient-capability-invocation-broker.test.ts | 10 +++- .../client-capability-protocol.test.ts | 34 ++++++++++- .../__tests__/client-capability-uds.test.ts | 16 +++++- .../execution-host-continuation.test.ts | 2 +- .../execution-model-composition.test.ts | 2 + .../src/__tests__/oauth-coordinator.test.ts | 2 +- .../__tests__/root-turn-coordinator.test.ts | 10 +++- .../session-continuity-coordinator.test.ts | 32 +++++++++++ .../src/adapter/session-projector.ts | 4 +- .../src/client/client-capability-channel.ts | 1 + .../src/protocol/client-capability.ts | 13 ++++- packages/runtime-host/src/protocol/index.ts | 4 +- .../src/protocol/session-continuity.ts | 8 ++- .../server/client-capability-coordinator.ts | 3 + .../server/session-continuity-coordinator.ts | 3 +- .../src/__tests__/computer-use-tools.test.ts | 3 + .../__tests__/computer-use-wait-for.test.ts | 10 +++- .../runtime-event-read-model.test.ts | 30 ++++++++++ .../tool-runtime-durable-boundary.test.ts | 47 +++++++++++++++ packages/runtime/src/computer-use-tools.ts | 52 +++++++++++++++-- packages/runtime/src/mcp-tools.ts | 5 ++ .../runtime/src/runtime-event-backfill.ts | 1 + .../runtime/src/runtime-event-read-model.ts | 1 + .../src/session-event-runtime-mapper.ts | 1 + packages/runtime/src/tool-runtime.ts | 57 ++++++++++++++----- .../__tests__/sqlite-runtime-store.test.ts | 30 ++++++++++ .../__tests__/live-turn-projection.test.ts | 15 +++++ packages/ui/src/live-turn-projection.ts | 2 +- packages/ui/src/materialize.ts | 2 +- 55 files changed, 559 insertions(+), 77 deletions(-) diff --git a/apps/desktop/src/main/__tests__/browser-tools.test.ts b/apps/desktop/src/main/__tests__/browser-tools.test.ts index 6ecce3c0e2..8af1becfe2 100644 --- a/apps/desktop/src/main/__tests__/browser-tools.test.ts +++ b/apps/desktop/src/main/__tests__/browser-tools.test.ts @@ -293,6 +293,7 @@ describe('browser tool execution', () => { ); assert.equal(resolved, 2); assert.deepEqual(result, { + outcome: 'success', content: [ { type: 'text', diff --git a/apps/desktop/src/main/__tests__/mcp-form-host-integration.test.ts b/apps/desktop/src/main/__tests__/mcp-form-host-integration.test.ts index cf797e6aa9..c41897a0b1 100644 --- a/apps/desktop/src/main/__tests__/mcp-form-host-integration.test.ts +++ b/apps/desktop/src/main/__tests__/mcp-form-host-integration.test.ts @@ -49,7 +49,7 @@ for (const action of ['accept', 'decline', 'cancel'] as const) { const answered = await host.answer(pending.requestId, action === 'accept' ? { action, values } : { action }); assert.equal(answered.ok, true, JSON.stringify(answered)); const settled = await result; - assert.deepEqual(settled.result, { content: [{ type: 'text', text: 'complete' }] }); + assert.deepEqual(settled.result, { outcome: 'success', content: [{ type: 'text', text: 'complete' }] }); assert.equal(server.calls.length, 2); assert.notEqual(server.calls[0]?.id, server.calls[1]?.id); assert.deepEqual(server.calls[1]?.params.arguments, server.calls[0]?.params.arguments); @@ -82,7 +82,7 @@ test('Desktop handles a second MCP question as a new canonical Host form', { tim assert.notEqual(second.requestId, first.requestId); assert.equal(server.calls.length, 2); assert.equal((await host.answer(second.requestId, { action: 'accept', values })).ok, true); - assert.deepEqual((await result).result, { content: [{ type: 'text', text: 'complete' }] }); + assert.deepEqual((await result).result, { outcome: 'success', content: [{ type: 'text', text: 'complete' }] }); assert.equal(server.calls.length, 3); assert.deepEqual(server.calls[1]?.params.inputResponses, { form: { action: 'decline' } }); assert.deepEqual(server.calls[2]?.params.inputResponses, { form: { action: 'accept', content: values } }); diff --git a/apps/desktop/src/main/__tests__/mcp-runtime-e2e.test.ts b/apps/desktop/src/main/__tests__/mcp-runtime-e2e.test.ts index 7d14c551ed..ff5837712e 100644 --- a/apps/desktop/src/main/__tests__/mcp-runtime-e2e.test.ts +++ b/apps/desktop/src/main/__tests__/mcp-runtime-e2e.test.ts @@ -156,6 +156,7 @@ test('MCP tools stay bound to the connection generation that advertised them', a }, ), { + outcome: 'success', content: [{ type: 'text', text: 'annotated:desktop-capability' }], }, ); @@ -185,6 +186,7 @@ test('MCP tools stay bound to the connection generation that advertised them', a }, ), { + outcome: 'success', content: [ { type: 'text', text: 'desktop-capability' }, { type: 'text', text: '{"structuredContent":{"echoed":"desktop-capability"}}' }, diff --git a/apps/desktop/src/main/__tests__/runtime-host-desktop-candidate.test.ts b/apps/desktop/src/main/__tests__/runtime-host-desktop-candidate.test.ts index 7509f10fd9..0c491ef80f 100644 --- a/apps/desktop/src/main/__tests__/runtime-host-desktop-candidate.test.ts +++ b/apps/desktop/src/main/__tests__/runtime-host-desktop-candidate.test.ts @@ -613,12 +613,14 @@ test('refreshes native capabilities with a new immutable provider snapshot', asy }; assert.deepEqual(await host.invokeCapability(frame), { + outcome: 'success', content: [{ type: 'text', text: 'old' }], }); implementation = 'new'; await controls?.refreshClientCapabilities(); assert.equal(host.capabilityRegistrations, 2); assert.deepEqual(await host.invokeCapability(frame), { + outcome: 'success', content: [{ type: 'text', text: 'new' }], }); @@ -797,7 +799,7 @@ test('isolates an invalid dynamic MCP tool without dropping the Host connection' serverId: 'desktop_mcp', toolName: 'healthy_mcp', }), - { content: [{ type: 'text', text: 'healthy' }] }, + { outcome: 'success', content: [{ type: 'text', text: 'healthy' }] }, ); await candidate.close(); diff --git a/apps/desktop/src/main/__tests__/runtime-host-native-capabilities.test.ts b/apps/desktop/src/main/__tests__/runtime-host-native-capabilities.test.ts index 6bca312df3..6af1ddbf37 100644 --- a/apps/desktop/src/main/__tests__/runtime-host-native-capabilities.test.ts +++ b/apps/desktop/src/main/__tests__/runtime-host-native-capabilities.test.ts @@ -272,7 +272,7 @@ test('forwards JSON Schema native capability arguments to the MCP authority', as arguments: {}, }), ), - { content: [{ type: 'text', text: 'server result' }] }, + { outcome: 'success', content: [{ type: 'text', text: 'server result' }] }, ); assert.deepEqual(receivedArguments, {}); }); @@ -670,7 +670,7 @@ test('validates before admission and invokes the exact offered tool with Host co admitted = true; }, ); - assert.deepEqual(result, { content: [{ type: 'text', text: 'Loaded' }] }); + assert.deepEqual(result, { outcome: 'success', content: [{ type: 'text', text: 'Loaded' }] }); assert.deepEqual(received, { args: { url: 'https://example.com/path' }, context: { @@ -809,6 +809,7 @@ test('projects Computer Use screenshots and releases all native resources for a const completed = await call(provider, computerFrame({ sessionId: 'completed-session', arguments: {} })); assert.deepEqual(completed, { + outcome: 'success', content: [ { type: 'text', text: 'captured' }, { type: 'image', data: 'aW1hZ2U=', mimeType: 'image/png' }, @@ -838,6 +839,19 @@ test('projects Computer Use screenshots and releases all native resources for a await assert.rejects(() => call(provider, capabilityFrame()), /provider is closed/u); }); +test('projects native Computer Use refusal separately from its model text', async () => { + const backend = computerBackend(); + backend.preflight = async () => ({ accessibility: false, screenRecording: true }); + const provider = createDesktopNativeCapabilityProvider({ + browserTools: [], resolveBrowserUrl: () => 'https://example.com/', releaseBrowserSession() {}, + computerUseTools: buildComputerUseTools({ backend }), releaseDesktopInteractionSession() {}, + }); + const result = await call(provider, computerFrame({ arguments: { action: 'wait', duration: 0.001 } })); + assert.equal(result.outcome, 'error'); + assert.match(result.content[0]?.type === 'text' ? result.content[0].text : '', /permission_missing/); + await provider.close(); +}); + test('does not advertise unavailable capability groups or dispatch unknown identities', async () => { const provider = createDesktopNativeCapabilityProvider({ browserTools: [tool('browser_snapshot', z.object({}), async () => 'ok')], @@ -900,7 +914,7 @@ test('dispatches through the same immutable tool snapshot it advertised', async toolName: 'old_tool', }), ), - { content: [{ type: 'text', text: 'old implementation' }] }, + { outcome: 'success', content: [{ type: 'text', text: 'old implementation' }] }, ); await assert.rejects( () => @@ -965,7 +979,7 @@ test('chunks a dynamic capability group beyond the single-offer tool limit', asy arguments: {}, }), ), - { content: [{ type: 'text', text: 'tool-64' }] }, + { outcome: 'success', content: [{ type: 'text', text: 'tool-64' }] }, ); await provider.close(); }); @@ -1140,7 +1154,7 @@ test('publishes identified tools under their real normalized MCP identity', asyn arguments: {}, }), ), - { content: [{ type: 'text', text: 'echo result' }] }, + { outcome: 'success', content: [{ type: 'text', text: 'echo result' }] }, ); assert.deepEqual( await call( @@ -1152,7 +1166,7 @@ test('publishes identified tools under their real normalized MCP identity', asyn arguments: {}, }), ), - { content: [{ type: 'text', text: 'run result' }] }, + { outcome: 'success', content: [{ type: 'text', text: 'run result' }] }, ); await provider.close(); }); diff --git a/apps/desktop/src/main/computer-use-real-model-policy.ts b/apps/desktop/src/main/computer-use-real-model-policy.ts index ef396aeceb..fcf6c7a3b9 100644 --- a/apps/desktop/src/main/computer-use-real-model-policy.ts +++ b/apps/desktop/src/main/computer-use-real-model-policy.ts @@ -109,12 +109,14 @@ export function applyComputerUseRealModelPolicy( totalActions += 1; if (totalActions > policy.maxTotalActions) { return { + outcome: 'error', text: 'maka_computer failed: total_action_budget_exceeded', error: 'total_action_budget_exceeded', }; } if (!allowed.has(action)) { return { + outcome: 'error', text: `maka_computer.${action} failed: unsupported_action_policy`, error: 'unsupported_action_policy', }; @@ -125,6 +127,7 @@ export function applyComputerUseRealModelPolicy( && (typeof app !== 'string' || !allowedApps.has(app)) ) { return { + outcome: 'error', text: `maka_computer.${action} failed: target_policy_mismatch`, error: 'target_policy_mismatch', }; @@ -142,6 +145,7 @@ export function applyComputerUseRealModelPolicy( ) ) { return { + outcome: 'error', text: `maka_computer.${action} failed: target_policy_mismatch`, error: 'target_policy_mismatch', }; @@ -150,6 +154,7 @@ export function applyComputerUseRealModelPolicy( actionCounts.set(action, actionCount); if (actionCount > (policy.maxActionCounts[action] ?? 0)) { return { + outcome: 'error', text: `maka_computer.${action} failed: action_budget_exceeded`, error: 'action_budget_exceeded', }; diff --git a/apps/desktop/src/main/runtime-host-native-capabilities.ts b/apps/desktop/src/main/runtime-host-native-capabilities.ts index fdaf0fe129..a68909f3d5 100644 --- a/apps/desktop/src/main/runtime-host-native-capabilities.ts +++ b/apps/desktop/src/main/runtime-host-native-capabilities.ts @@ -20,6 +20,8 @@ import { Buffer } from "node:buffer"; import type { ComputerUseToolSet } from '@maka/runtime/computer-use-tools'; import type { MakaTool } from '@maka/runtime/tool-runtime'; +import { coerceResultContent, deriveToolResultStatus } from '@maka/runtime/tool-runtime'; +import { requireToolCallOutcome } from '@maka/core/tool-result-status'; import { createOAuthPresentationClientProvider, type ClientCapabilityProvider, @@ -730,6 +732,9 @@ async function projectToolResult( input: unknown, output: unknown, ): Promise { + const outcome = tool.resultOutcome + ? requireToolCallOutcome(tool.resultOutcome(output)) + : deriveToolResultStatus(coerceResultContent(output), output); const modelOutput = tool.toModelOutput ? await tool.toModelOutput({ toolCallId, @@ -739,24 +744,25 @@ async function projectToolResult( : undefined; if (!modelOutput) { return typeof output === "string" - ? { content: [{ type: "text", text: output }] } - : { content: [], structuredContent: output }; + ? { outcome, content: [{ type: "text", text: output }] } + : { outcome, content: [], structuredContent: output }; } switch (modelOutput.type) { case "text": case "error-text": - return { content: [{ type: "text", text: modelOutput.value }] }; + return { outcome, content: [{ type: "text", text: modelOutput.value }] }; case "json": case "error-json": - return { content: [], structuredContent: modelOutput.value }; + return { outcome, content: [], structuredContent: modelOutput.value }; case "execution-denied": return { + outcome, content: [ { type: "text", text: modelOutput.reason ?? "Execution denied" }, ], }; case "content": - return { content: modelOutput.value.map(projectContentPart) }; + return { outcome, content: modelOutput.value.map(projectContentPart) }; } } diff --git a/packages/cli/src/__tests__/acp-session-event-mapper.test.ts b/packages/cli/src/__tests__/acp-session-event-mapper.test.ts index 14c98772a9..dfa42b914b 100644 --- a/packages/cli/src/__tests__/acp-session-event-mapper.test.ts +++ b/packages/cli/src/__tests__/acp-session-event-mapper.test.ts @@ -275,6 +275,32 @@ describe('ACP Session event mapper', () => { assert.equal(notifications.length, count); }); + test('updates ACP host status when only the explicit outcome changes', async () => { + const notifications: SessionNotification[] = []; + const mapper = eventMapper(notifications); + for (const outcome of ['error', 'aborted'] as const) { + await mapper.accept( + event({ + type: 'tool_result', + toolUseId: 'tool', + contentOmitted: true, + isError: true, + outcome, + content: { kind: 'text', text: '' }, + }), + ); + } + const updates = notifications.map(toolUpdate); + assert.deepEqual( + updates.map((update) => (update._meta?.maka as { hostStatus?: string })?.hostStatus), + ['errored', 'interrupted'], + ); + assert.deepEqual( + updates.map((update) => update.status), + ['failed', 'failed'], + ); + }); + test('result before start creates one terminal card and late start only fills identity', async () => { const notifications: SessionNotification[] = []; const mapper = eventMapper(notifications); diff --git a/packages/cli/src/__tests__/mcp-form-host-integration.test.ts b/packages/cli/src/__tests__/mcp-form-host-integration.test.ts index a8e495f734..6a076ce757 100644 --- a/packages/cli/src/__tests__/mcp-form-host-integration.test.ts +++ b/packages/cli/src/__tests__/mcp-form-host-integration.test.ts @@ -56,7 +56,10 @@ for (const action of ['accept', 'decline', 'cancel'] as const) { ); assert.equal(answered.ok, true, JSON.stringify(answered)); const settled = await result; - assert.deepEqual(settled.result, { content: [{ type: 'text', text: 'complete' }] }); + assert.deepEqual(settled.result, { + outcome: 'success', + content: [{ type: 'text', text: 'complete' }], + }); assert.equal(server.calls.length, 2); assert.notEqual(server.calls[0]?.id, server.calls[1]?.id); assert.deepEqual(server.calls[1]?.params.arguments, server.calls[0]?.params.arguments); @@ -92,7 +95,10 @@ test('TUI handles a second MCP question as a new canonical Host form', { assert.notEqual(second.requestId, first.requestId); assert.equal(server.calls.length, 2); assert.equal((await host.answer(second.requestId, { action: 'accept', values })).ok, true); - assert.deepEqual((await result).result, { content: [{ type: 'text', text: 'complete' }] }); + assert.deepEqual((await result).result, { + outcome: 'success', + content: [{ type: 'text', text: 'complete' }], + }); assert.equal(server.calls.length, 3); assert.deepEqual(server.calls[1]?.params.inputResponses, { form: { action: 'decline' } }); assert.deepEqual(server.calls[2]?.params.inputResponses, { diff --git a/packages/cli/src/__tests__/runtime-host-capability-provider-command.test.ts b/packages/cli/src/__tests__/runtime-host-capability-provider-command.test.ts index e5cf430c3e..a4bc68a985 100644 --- a/packages/cli/src/__tests__/runtime-host-capability-provider-command.test.ts +++ b/packages/cli/src/__tests__/runtime-host-capability-provider-command.test.ts @@ -152,7 +152,10 @@ test('MCP capability publication freezes an accepted callable tool snapshot', as requestInteraction: async () => assert.fail('Unexpected provider interaction'), }, ); - assert.deepEqual(result, { content: [{ type: 'text', text: '{"path":"README.md"}' }] }); + assert.deepEqual(result, { + outcome: 'success', + content: [{ type: 'text', text: '{"path":"README.md"}' }], + }); }); test('MCP capability publication packs tools across server boundaries', () => { diff --git a/packages/cli/src/__tests__/tui-mcp-local-integration.test.ts b/packages/cli/src/__tests__/tui-mcp-local-integration.test.ts index aab5ec79ff..438dca61a2 100644 --- a/packages/cli/src/__tests__/tui-mcp-local-integration.test.ts +++ b/packages/cli/src/__tests__/tui-mcp-local-integration.test.ts @@ -125,6 +125,7 @@ test('local TUI discovers, publishes, invokes, republishes, and closes one MCP c assert.deepEqual( await firstTool.impl({ value: 'before reconnect' }, toolContext(hostRoot, 'call-1')), { + outcome: 'success', content: [{ type: 'text', text: 'before reconnect' }], structuredContent: { echoed: 'before reconnect' }, }, @@ -148,6 +149,7 @@ test('local TUI discovers, publishes, invokes, republishes, and closes one MCP c assert.deepEqual( await replacementTool.impl({ value: 'after reconnect' }, toolContext(hostRoot, 'call-2')), { + outcome: 'success', content: [{ type: 'text', text: 'after reconnect' }], structuredContent: { echoed: 'after reconnect' }, }, diff --git a/packages/cli/src/__tests__/tui-mcp-remote-integration.test.ts b/packages/cli/src/__tests__/tui-mcp-remote-integration.test.ts index b643e24b7b..6a8ab19860 100644 --- a/packages/cli/src/__tests__/tui-mcp-remote-integration.test.ts +++ b/packages/cli/src/__tests__/tui-mcp-remote-integration.test.ts @@ -428,7 +428,10 @@ function dummyProvider(id: string) { ], }, ], - call: async () => ({ content: [{ type: 'text' as const, text: id }] }), + call: async () => ({ + outcome: 'success' as const, + content: [{ type: 'text' as const, text: id }], + }), }; } diff --git a/packages/cli/src/acp/tool-event-mapper.ts b/packages/cli/src/acp/tool-event-mapper.ts index 9559ade323..1b539ceb6b 100644 --- a/packages/cli/src/acp/tool-event-mapper.ts +++ b/packages/cli/src/acp/tool-event-mapper.ts @@ -33,6 +33,7 @@ import { import type { StoredMessage } from '@maka/core/session'; import { projectToolArgsPreview } from '@maka/core/tool-quiet-preview'; import { toolResultActivityStatus } from '@maka/core/tool-result-status'; +import type { ToolCallOutcome } from '@maka/core/tool-result-status'; import { BoundedChunkBuffer } from '../bounded-chunk-buffer.js'; import { formatToolResultContent } from '../pi-transcript-format.js'; @@ -140,6 +141,7 @@ export class AcpToolEventMapper { event.content, event.durationMs, event.contentOmitted === true, + event.outcome, ); return; } @@ -157,6 +159,7 @@ export class AcpToolEventMapper { message.content, message.durationMs, false, + message.outcome, ); } } @@ -226,9 +229,11 @@ export class AcpToolEventMapper { result: ToolResultContent, durationMs: number | undefined, omitted: boolean, + outcome?: ToolCallOutcome, ): Promise { const resultDigest = digestValue({ isError, + outcome, result: omitted ? null : result, durationMs, omitted, @@ -236,7 +241,7 @@ export class AcpToolEventMapper { if (tool.resultDigest === resultDigest) return; if (omitted) tool.resultAnnounced = true; tool.terminal = true; - const hostStatus = toolResultActivityStatus(isError, omitted ? undefined : result); + const hostStatus = toolResultActivityStatus(isError, omitted ? undefined : result, outcome); tool.status = hostStatus === 'completed' ? 'completed' : 'failed'; tool.meta.hostStatus = hostStatus; if (durationMs !== undefined) tool.meta.durationMs = durationMs; diff --git a/packages/cli/src/mcp-capability-provider.ts b/packages/cli/src/mcp-capability-provider.ts index ca0849514d..b4d7e7c601 100644 --- a/packages/cli/src/mcp-capability-provider.ts +++ b/packages/cli/src/mcp-capability-provider.ts @@ -128,6 +128,7 @@ function projectMcpTool(tool: McpToolDescriptor, wireServerId: string) { function projectMcpResult(result: McpCallResult): ClientCapabilityCallResult { return { + outcome: 'success', content: result.content.map((block) => structuredClone(block)), ...(result.structuredContent === undefined ? {} diff --git a/packages/cli/src/pi-transcript.ts b/packages/cli/src/pi-transcript.ts index 90528fd184..5e745ea26d 100644 --- a/packages/cli/src/pi-transcript.ts +++ b/packages/cli/src/pi-transcript.ts @@ -870,7 +870,7 @@ export function applyMakaSessionEventToTranscript( } if (tool) { if (tool.suppressed) unsuppressToolAtTail(state, tool); - tool.callStatus = toolResultActivityStatus(event.isError, event.content); + tool.callStatus = toolResultActivityStatus(event.isError, event.content, event.outcome); if (shellRun) { if (tool.toolName === 'Bash') { applyShellRunResult(tool, shellRun); @@ -898,7 +898,7 @@ export function applyMakaSessionEventToTranscript( ...(!event.contentOmitted ? { result: event.content } : {}), resultVersion: event.contentOmitted ? 0 : 1, durationMs: event.durationMs, - callStatus: toolResultActivityStatus(event.isError, event.content), + callStatus: toolResultActivityStatus(event.isError, event.content, event.outcome), expanded: state.expandAllTools, }); } @@ -1149,7 +1149,7 @@ function storedToolToTranscriptEntry( resultVersion: result ? 1 : 0, ...(result?.durationMs !== undefined ? { durationMs: result.durationMs } : {}), callStatus: result - ? toolResultActivityStatus(result.isError, result.content) + ? toolResultActivityStatus(result.isError, result.content, result.outcome) : unfinishedToolActivityStatus(turnStatus), expanded: false, }; diff --git a/packages/core/src/__tests__/tool-result-record-schema.test.ts b/packages/core/src/__tests__/tool-result-record-schema.test.ts index cb0b4be59d..863ba66d16 100644 --- a/packages/core/src/__tests__/tool-result-record-schema.test.ts +++ b/packages/core/src/__tests__/tool-result-record-schema.test.ts @@ -324,6 +324,18 @@ function storedToolResult(content: unknown) { }; } +test('explicit tool outcome survives both decoders and old rows remain readable', () => { + const legacy = storedToolResult({ kind: 'text', text: 'old' }); + assert.deepEqual(decodePersistedMessage(legacy), legacy); + for (const outcome of ['success', 'error', 'aborted'] as const) { + const row = { ...legacy, isError: outcome !== 'success', outcome }; + assert.deepEqual(decodeCanonicalMessage(row), row); + assert.deepEqual(decodePersistedMessage(row), row); + assert.throws(() => decodeCanonicalMessage({ ...row, isError: outcome === 'success' })); + } + assert.throws(() => decodePersistedMessage({ ...legacy, outcome: 'failed' })); +}); + function decodePersistedMessage(value: unknown): StoredMessage { return decodeStoredMessage(markPersisted(value)); } diff --git a/packages/core/src/events.ts b/packages/core/src/events.ts index cd12200618..024b98cf03 100644 --- a/packages/core/src/events.ts +++ b/packages/core/src/events.ts @@ -776,6 +776,8 @@ export interface ToolResultEvent extends BaseEvent, ToolActivityIdentity { /** The transport omitted durable result content; consumers must not treat the placeholder as authoritative. */ contentOmitted?: true; isError: boolean; + /** Explicit call outcome; absent on legacy and non-migrated tools. */ + outcome?: import('./tool-result-status.js').ToolCallOutcome; content: ToolResultContent; durationMs?: number; } diff --git a/packages/core/src/mcp.ts b/packages/core/src/mcp.ts index 2cc941ed73..1c15801d35 100644 --- a/packages/core/src/mcp.ts +++ b/packages/core/src/mcp.ts @@ -331,6 +331,8 @@ export type McpContentBlock = export interface McpCallResult { content: McpContentBlock[]; structuredContent?: unknown; + /** Validated Client Capability metadata; ordinary MCP calls do not set it. */ + outcome?: import('./tool-result-status.js').ToolCallOutcome; } export interface McpTestResult { diff --git a/packages/core/src/runtime-event.ts b/packages/core/src/runtime-event.ts index 77643ee0c4..b213e8e19e 100644 --- a/packages/core/src/runtime-event.ts +++ b/packages/core/src/runtime-event.ts @@ -34,6 +34,7 @@ */ import { isWorkHubActionReceipt, type WorkHubActionReceipt } from './workhub-action-result.js'; +import { isToolCallOutcome } from './tool-result-status.js'; import { isModelRetryDecision, type ModelRetryDecision } from './model-failure.js'; import { @@ -226,6 +227,7 @@ export interface RuntimeEventFunctionResponseContent { name: string; result: unknown; isError?: boolean; + outcome?: import('./tool-result-status.js').ToolCallOutcome; providerExecuted?: boolean; /** Raw provider result retained for provider-native replay; never rendered directly. */ providerOutput?: unknown; @@ -756,7 +758,7 @@ const FUNCTION_CALL_CONTENT_SHAPE = defineObjectShape()( ['kind', 'id', 'name', 'result'], - ['isError', 'providerExecuted', 'providerOutput', 'modelProjection'], + ['isError', 'outcome', 'providerExecuted', 'providerOutput', 'modelProjection'], ); const ERROR_CONTENT_SHAPE = defineObjectShape()( ['kind', 'message'], @@ -1112,6 +1114,9 @@ function isRuntimeEventContent(value: unknown): value is RuntimeEventContent { typeof value.name === 'string' && Object.hasOwn(value, 'result') && (value.isError === undefined || typeof value.isError === 'boolean') && + (value.outcome === undefined || + (isToolCallOutcome(value.outcome) && + (value.outcome !== 'success') === (value.isError === true))) && (value.providerExecuted === undefined || typeof value.providerExecuted === 'boolean') && (value.modelProjection === undefined || decodesDurableToolResultProjection(value.modelProjection)) diff --git a/packages/core/src/session.ts b/packages/core/src/session.ts index 924704f44b..19c029d91c 100644 --- a/packages/core/src/session.ts +++ b/packages/core/src/session.ts @@ -19,6 +19,7 @@ import { isWorkHubActionReceipt, type WorkHubActionReceipt } from './workhub-action-result.js'; import { isExecutorId } from './executor-id.js'; +import { isToolCallOutcome } from './tool-result-status.js'; import { MODEL_FAILURE_MESSAGE_MAX_BYTES, @@ -901,6 +902,7 @@ export interface ToolResultMessage { /** Matches ToolCallMessage.id. */ toolUseId: string; isError: boolean; + outcome?: import('./tool-result-status.js').ToolCallOutcome; content: ToolResultContent; providerExecuted?: boolean; /** Raw provider result retained only for provider-native replay. */ @@ -1306,6 +1308,7 @@ const TOOL_RESULT_MESSAGE_SHAPE = defineObjectShape()( ['type', 'id', 'turnId', 'ts', 'toolUseId', 'isError', 'content'], [ 'durationMs', + 'outcome', 'providerExecuted', 'providerOutput', 'origin', @@ -1611,6 +1614,9 @@ function decodeMessage( hasMessageEnvelope(message, true) && typeof message.toolUseId === 'string' && typeof message.isError === 'boolean' && + (message.outcome === undefined || + (isToolCallOutcome(message.outcome) && + (message.outcome !== 'success') === message.isError)) && (message.providerExecuted === undefined || typeof message.providerExecuted === 'boolean') && isOptionalFiniteDuration(message.durationMs) && isToolActivityIdentity(message) diff --git a/packages/core/src/tool-result-status.ts b/packages/core/src/tool-result-status.ts index 5c06c9404d..345354c7c7 100644 --- a/packages/core/src/tool-result-status.ts +++ b/packages/core/src/tool-result-status.ts @@ -25,6 +25,16 @@ import type { ToolResultContent } from './events.js'; import type { TurnStatus } from './session.js'; export type SettledToolActivityStatus = 'completed' | 'errored' | 'interrupted'; +export type ToolCallOutcome = 'success' | 'error' | 'aborted'; + +export function isToolCallOutcome(value: unknown): value is ToolCallOutcome { + return value === 'success' || value === 'error' || value === 'aborted'; +} + +export function requireToolCallOutcome(value: unknown): ToolCallOutcome { + if (!isToolCallOutcome(value)) throw new Error('Invalid tool call outcome'); + return value; +} /** A call that has started and has not settled. */ export type InFlightToolActivityStatus = 'running'; @@ -79,7 +89,11 @@ export function isCancelledToolResultContent(content: ToolResultContent | undefi export function toolResultActivityStatus( isError: boolean, content: ToolResultContent | undefined, + outcome?: ToolCallOutcome, ): SettledToolActivityStatus { + if (outcome !== undefined) { + return outcome === 'success' ? 'completed' : outcome === 'aborted' ? 'interrupted' : 'errored'; + } if (!isError) return 'completed'; // Failed cancel (user stop / kill) — not a tool failure banner. if (isCancelledToolResultContent(content)) return 'interrupted'; diff --git a/packages/runtime-host/src/__tests__/authenticated-websocket.test.ts b/packages/runtime-host/src/__tests__/authenticated-websocket.test.ts index dfc022e4b2..4f0b845292 100644 --- a/packages/runtime-host/src/__tests__/authenticated-websocket.test.ts +++ b/packages/runtime-host/src/__tests__/authenticated-websocket.test.ts @@ -872,7 +872,10 @@ test('an unbound WebSocket credential cannot claim an existing bound Client iden ], }, ], - call: async () => ({ content: [{ type: 'text', text: 'bound' }] }), + call: async () => ({ + outcome: 'success', + content: [{ type: 'text', text: 'bound' }], + }), }); assert.deepEqual( diff --git a/packages/runtime-host/src/__tests__/client-capability-admission-integration.test.ts b/packages/runtime-host/src/__tests__/client-capability-admission-integration.test.ts index e7fdd58d98..979b9e8798 100644 --- a/packages/runtime-host/src/__tests__/client-capability-admission-integration.test.ts +++ b/packages/runtime-host/src/__tests__/client-capability-admission-integration.test.ts @@ -106,7 +106,10 @@ test('cancels managed approval owners and joiners with the canonical provider id connection.accept({ kind: 'client.capability.result', invocationId: frame.invocationId, - result: { content: [{ type: 'text', text: 'snapshot' }] }, + result: { + outcome: 'success', + content: [{ type: 'text', text: 'snapshot' }], + }, }); } }, @@ -256,7 +259,10 @@ test('cancels managed approval owners and joiners with the canonical provider id ...retryRequest.request.target, }); assert.equal(grant?.providerId, providerId); - assert.deepEqual(settlement.result, { content: [{ type: 'text', text: 'snapshot' }] }); + assert.deepEqual(settlement.result, { + outcome: 'success', + content: [{ type: 'text', text: 'snapshot' }], + }); assert.deepEqual(order, ['accepted', 'accepted', 'accepted', 'approved', 'T1', 'admitted']); } finally { snapshot?.release(); diff --git a/packages/runtime-host/src/__tests__/client-capability-channel.test.ts b/packages/runtime-host/src/__tests__/client-capability-channel.test.ts index 337c27e998..be4195152c 100644 --- a/packages/runtime-host/src/__tests__/client-capability-channel.test.ts +++ b/packages/runtime-host/src/__tests__/client-capability-channel.test.ts @@ -42,7 +42,7 @@ test('Client Capability channel closes a provider after its final registration i ], }, ], - call: async () => ({ content: [] }), + call: async () => ({ outcome: 'success', content: [] }), close: () => { closeCalls += 1; }, @@ -142,7 +142,11 @@ test('Client Capability channel runs a self-described Host service through admis { kind: 'client.capability.result', invocationId: 'service_invocation', - result: { content: [], structuredContent: { kind: 'presented' } }, + result: { + outcome: 'success', + content: [], + structuredContent: { kind: 'presented' }, + }, }, ]); channel.close(new Error('test complete')); @@ -171,7 +175,7 @@ test('Client Capability channel rejects Host paths before invoking a path-isolat ], call: async () => { callCount += 1; - return { content: [] }; + return { outcome: 'success', content: [] }; }, }; const channel = new ClientCapabilityChannel({ @@ -234,7 +238,7 @@ test('Client Capability channel forwards admitted tool progress before the resul await options.accept({ kind: 'none' }); options.progress?.(1, 3); options.progress?.(2, 3); - return { content: [] }; + return { outcome: 'success', content: [] }; }, }; channel = new ClientCapabilityChannel({ @@ -296,7 +300,7 @@ test('Client Capability channel forwards admitted tool progress before the resul { kind: 'client.capability.result', invocationId: 'progress-invocation', - result: { content: [] }, + result: { outcome: 'success', content: [] }, }, ]); channel.close(new Error('test complete')); @@ -335,8 +339,14 @@ test('Client Capability channel correlates one admitted nested form before the f }, ], }); - assert.deepEqual(answer, { action: 'accept', values: { target: 'staging' } }); - return { content: [{ type: 'text', text: 'deployed' }] }; + assert.deepEqual(answer, { + action: 'accept', + values: { target: 'staging' }, + }); + return { + outcome: 'success', + content: [{ type: 'text', text: 'deployed' }], + }; }, }; channel = new ClientCapabilityChannel({ @@ -396,9 +406,15 @@ test('Client Capability channel correlates one admitted nested form before the f assert.deepEqual(written.at(-1), { kind: 'client.capability.result', invocationId: 'nested-form', - result: { content: [{ type: 'text', text: 'deployed' }] }, + result: { + outcome: 'success', + content: [{ type: 'text', text: 'deployed' }], + }, + }); + channel.accept({ + kind: 'client.capability.release', + invocationId: 'nested-form', }); - channel.accept({ kind: 'client.capability.release', invocationId: 'nested-form' }); channel.close(new Error('test complete')); }); @@ -429,7 +445,7 @@ test('Client Capability release rejects a pending nested form', async () => { requester: { name: 'deploy' }, fields: [{ kind: 'boolean', name: 'confirm', label: 'Confirm', required: true }], }); - return { content: [] }; + return { outcome: 'success', content: [] }; } catch (error) { observedError = error; throw error; diff --git a/packages/runtime-host/src/__tests__/client-capability-coordinator.test.ts b/packages/runtime-host/src/__tests__/client-capability-coordinator.test.ts index b6156c3eec..ed9c183991 100644 --- a/packages/runtime-host/src/__tests__/client-capability-coordinator.test.ts +++ b/packages/runtime-host/src/__tests__/client-capability-coordinator.test.ts @@ -22,7 +22,7 @@ import { describe, test } from 'node:test'; import { createManagedExecutionBoundary } from '@maka/core/sandbox-boundary'; import { createWorkspaceWritePermissionProfile } from '@maka/core/permission-profile'; import { ToolOutcomeUnknownError } from '@maka/core/events'; -import type { McpCallResult } from '@maka/core/mcp'; +import type { ClientCapabilityCallResult } from '../protocol/index.js'; import type { ClientCapabilityAdmissionEvidence, ClientCapabilityCallFrame, @@ -2003,7 +2003,11 @@ test('Host services stay bound to the explicitly initiating Client connection', connection.accept({ kind: 'client.capability.result', invocationId: frame.invocationId, - result: { content: [], structuredContent: { kind: 'presented' } }, + result: { + outcome: 'success', + content: [], + structuredContent: { kind: 'presented' }, + }, }); } }, @@ -2253,8 +2257,8 @@ function toolAt( return snapshot?.tools[index]; } -function textResult(text: string): McpCallResult { - return { content: [{ type: 'text', text }] }; +function textResult(text: string): ClientCapabilityCallResult { + return { outcome: 'success', content: [{ type: 'text', text }] }; } function isRecord(value: unknown): value is Record { diff --git a/packages/runtime-host/src/__tests__/client-capability-interaction-broker.test.ts b/packages/runtime-host/src/__tests__/client-capability-interaction-broker.test.ts index 31e23f4d73..6166737bf3 100644 --- a/packages/runtime-host/src/__tests__/client-capability-interaction-broker.test.ts +++ b/packages/runtime-host/src/__tests__/client-capability-interaction-broker.test.ts @@ -123,7 +123,10 @@ test('Client Capability accepts a final result while interaction delivery is sti broker.accept('connection-a', { kind: 'client.capability.result', invocationId: frame.invocationId, - result: { content: [{ type: 'text', text: 'deployed' }] }, + result: { + outcome: 'success', + content: [{ type: 'text', text: 'deployed' }], + }, }); await flush(); }, @@ -160,7 +163,10 @@ test('Client Capability accepts a final result while interaction delivery is sti }, }); - assert.deepEqual(await result, { content: [{ type: 'text', text: 'deployed' }] }); + assert.deepEqual(await result, { + outcome: 'success', + content: [{ type: 'text', text: 'deployed' }], + }); broker.close(); }); diff --git a/packages/runtime-host/src/__tests__/client-capability-invocation-broker.test.ts b/packages/runtime-host/src/__tests__/client-capability-invocation-broker.test.ts index 01bab875c9..c74297c0b5 100644 --- a/packages/runtime-host/src/__tests__/client-capability-invocation-broker.test.ts +++ b/packages/runtime-host/src/__tests__/client-capability-invocation-broker.test.ts @@ -71,7 +71,10 @@ describe('ClientCapabilityInvocationBroker', () => { broker.accept('connection-a', { kind: 'client.capability.result', invocationId: frame.invocationId, - result: { content: [{ type: 'text', text: 'ok' }] }, + result: { + outcome: 'success', + content: [{ type: 'text', text: 'ok' }], + }, }), ); } @@ -88,7 +91,10 @@ describe('ClientCapabilityInvocationBroker', () => { false, ); - assert.deepEqual(await prepared.admit(), { content: [{ type: 'text', text: 'ok' }] }); + assert.deepEqual(await prepared.admit(), { + outcome: 'success', + content: [{ type: 'text', text: 'ok' }], + }); assert.equal(sent.filter((frame) => frame.kind === 'client.capability.admitted').length, 1); broker.close(); }); diff --git a/packages/runtime-host/src/__tests__/client-capability-protocol.test.ts b/packages/runtime-host/src/__tests__/client-capability-protocol.test.ts index 332829d622..9297cf998c 100644 --- a/packages/runtime-host/src/__tests__/client-capability-protocol.test.ts +++ b/packages/runtime-host/src/__tests__/client-capability-protocol.test.ts @@ -31,6 +31,36 @@ import { } from '../protocol/index.js'; describe('Client Capability protocol', () => { + test('requires a tri-state result outside business structuredContent', () => { + for (const outcome of ['success', 'error', 'aborted'] as const) { + assert.deepEqual( + decodeClientCapabilityResult({ + outcome, + content: [], + structuredContent: { outcome: 'business' }, + }), + { + outcome, + content: [], + structuredContent: { outcome: 'business' }, + }, + ); + } + for (const outcome of [undefined, null, true, 'failed', 'SUCCESS']) { + assert.throws( + () => decodeClientCapabilityResult({ outcome, content: [] }), + RuntimeHostProtocolError, + ); + } + assert.throws( + () => + decodeClientCapabilityResult({ + content: [], + structuredContent: { outcome: 'error' }, + }), + RuntimeHostProtocolError, + ); + }); test('preserves opaque tool-call IDs while retaining identity bounds', () => { const frame = { kind: 'client.capability.call', @@ -619,6 +649,7 @@ describe('Client Capability protocol', () => { test('rejects non-canonical media data and invalid image MIME types', () => { assert.deepEqual( decodeClientCapabilityResult({ + outcome: 'success', content: [ { type: 'image', data: 'aGVsbG8=', mimeType: 'image/png' }, { type: 'audio', data: 'YQ==', mimeType: 'audio/wav' }, @@ -626,6 +657,7 @@ describe('Client Capability protocol', () => { ], }), { + outcome: 'success', content: [ { type: 'image', data: 'aGVsbG8=', mimeType: 'image/png' }, { type: 'audio', data: 'YQ==', mimeType: 'audio/wav' }, @@ -643,7 +675,7 @@ describe('Client Capability protocol', () => { [{ type: 'image', data: 'YQ==', mimeType: 'image/png; charset=binary' }], ]) { assert.throws( - () => decodeClientCapabilityResult({ content }), + () => decodeClientCapabilityResult({ outcome: 'success', content }), (error: unknown) => error instanceof RuntimeHostProtocolError, ); } diff --git a/packages/runtime-host/src/__tests__/client-capability-uds.test.ts b/packages/runtime-host/src/__tests__/client-capability-uds.test.ts index 737a9390d9..1c9c87eb09 100644 --- a/packages/runtime-host/src/__tests__/client-capability-uds.test.ts +++ b/packages/runtime-host/src/__tests__/client-capability-uds.test.ts @@ -170,6 +170,7 @@ test('unknown Client Capability loads, invokes, and rebinds after UDS reconnect' } await accept({ kind: 'none' }); return { + outcome: frame.arguments.prefix === 'failure' ? 'error' : 'success', content: [ { type: 'text', @@ -240,8 +241,14 @@ test('unknown Client Capability loads, invokes, and rebinds after UDS reconnect' const result = await tool.impl({ prefix: 'from-uds' }, toolContext); assert.deepEqual(result, { + outcome: 'success', content: [{ type: 'text', text: `from-uds:${largeValue}` }], }); + const failed = await tool.impl({ prefix: 'failure' }, toolContext); + assert.deepEqual(failed, { + outcome: 'error', + content: [{ type: 'text', text: `failure:${largeValue}` }], + }); await client.status(); await assert.rejects( async () => rejectedTool.impl({}, toolContext), @@ -277,7 +284,13 @@ test('unknown Client Capability loads, invokes, and rebinds after UDS reconnect' assert.equal(frame.toolCallId, toolCallId); await accept({ kind: 'none' }); return { - content: [{ type: 'text', text: `reconnected:${String(frame.arguments.prefix)}` }], + outcome: 'success', + content: [ + { + type: 'text', + text: `reconnected:${String(frame.arguments.prefix)}`, + }, + ], }; }, }); @@ -291,6 +304,7 @@ test('unknown Client Capability loads, invokes, and rebinds after UDS reconnect' ); assert.ok(reconnectedTool); assert.deepEqual(await reconnectedTool.impl({ prefix: 'from-uds' }, toolContext), { + outcome: 'success', content: [{ type: 'text', text: 'reconnected:from-uds' }], }); } finally { diff --git a/packages/runtime-host/src/__tests__/execution-host-continuation.test.ts b/packages/runtime-host/src/__tests__/execution-host-continuation.test.ts index b93591d3ec..81d8c13d0b 100644 --- a/packages/runtime-host/src/__tests__/execution-host-continuation.test.ts +++ b/packages/runtime-host/src/__tests__/execution-host-continuation.test.ts @@ -455,7 +455,7 @@ function resumeFixtureProvider( ], call: async (_frame, { accept }) => { await accept({ kind: 'none' }); - return { content: [{ type: 'text', text: 'ok' }] }; + return { outcome: 'success', content: [{ type: 'text', text: 'ok' }] }; }, }; } diff --git a/packages/runtime-host/src/__tests__/execution-model-composition.test.ts b/packages/runtime-host/src/__tests__/execution-model-composition.test.ts index 21949b0bfc..13d9ca8245 100644 --- a/packages/runtime-host/src/__tests__/execution-model-composition.test.ts +++ b/packages/runtime-host/src/__tests__/execution-model-composition.test.ts @@ -644,6 +644,7 @@ async function runPermissionUpdateHostRegression( kind: 'client.capability.result', invocationId: frame.invocationId, result: { + outcome: 'success', content: [{ type: 'text', text: CLIENT_CAPABILITY_RESULT_TEXT }], }, }); @@ -1973,6 +1974,7 @@ test('production backend preserves coordinator Client Capability semantics acros kind: 'client.capability.result', invocationId: frame.invocationId, result: { + outcome: 'success', content: [{ type: 'text', text: CLIENT_CAPABILITY_RESULT_TEXT }], }, }); diff --git a/packages/runtime-host/src/__tests__/oauth-coordinator.test.ts b/packages/runtime-host/src/__tests__/oauth-coordinator.test.ts index e8aca9060f..18b1918685 100644 --- a/packages/runtime-host/src/__tests__/oauth-coordinator.test.ts +++ b/packages/runtime-host/src/__tests__/oauth-coordinator.test.ts @@ -1181,7 +1181,7 @@ async function attachPresentation( connection.accept({ kind: 'client.capability.result', invocationId: frame.invocationId, - result: { content: [], structuredContent }, + result: { outcome: 'success', content: [], structuredContent }, }); }, }); diff --git a/packages/runtime-host/src/__tests__/root-turn-coordinator.test.ts b/packages/runtime-host/src/__tests__/root-turn-coordinator.test.ts index 7c0be71aa2..479ac074ce 100644 --- a/packages/runtime-host/src/__tests__/root-turn-coordinator.test.ts +++ b/packages/runtime-host/src/__tests__/root-turn-coordinator.test.ts @@ -4323,7 +4323,10 @@ async function assertSessionSuccessorCapabilityDegradation( previousProvider.accept({ kind: 'client.capability.result', invocationId: frame.invocationId, - result: { content: [{ type: 'text', text: 'previous' }] }, + result: { + outcome: 'success', + content: [{ type: 'text', text: 'previous' }], + }, }); }, }, @@ -4343,7 +4346,10 @@ async function assertSessionSuccessorCapabilityDegradation( followupProvider.accept({ kind: 'client.capability.result', invocationId: frame.invocationId, - result: { content: [{ type: 'text', text: 'followup' }] }, + result: { + outcome: 'success', + content: [{ type: 'text', text: 'followup' }], + }, }); }, }, diff --git a/packages/runtime-host/src/__tests__/session-continuity-coordinator.test.ts b/packages/runtime-host/src/__tests__/session-continuity-coordinator.test.ts index c584e4a3aa..c6e1631ded 100644 --- a/packages/runtime-host/src/__tests__/session-continuity-coordinator.test.ts +++ b/packages/runtime-host/src/__tests__/session-continuity-coordinator.test.ts @@ -2054,6 +2054,38 @@ test('tool_result clears retained tool_result_preview so a later open does not s coordinator.close(); }); +test('an aborted result remains interrupted when continuity omits its content', async () => { + const coordinator = new SessionContinuityCoordinator( + HOST_EPOCH, + async () => canonical(), + new SessionAdmissionGate(), + ); + const sink = new RecordingSink(); + const connection = attachTestConnection(coordinator, 'connection-aborted-result', sink); + const opened = await open(coordinator, 'connection-aborted-result'); + connection.activate(opened.subscriptionId); + await coordinator.acceptRuntimeEvent(SESSION_ID, 'run-1', { + type: 'tool_result', + id: 'result-1', + turnId: 'turn-1', + ts: 2, + toolUseId: 'tool-1', + isError: true, + outcome: 'aborted', + content: { kind: 'text', text: 'private result' }, + }); + await waitFor(() => sink.frames.length === 1); + const frame = sink.frames[0]; + assert.equal( + frame?.kind === 'subscription.session_event' && + frame.event.type === 'tool_result' && + frame.event.status, + 'interrupted', + ); + connection.abort(opened.subscriptionId); + coordinator.close(); +}); + test('publishes only the minimal sandbox failure reason from a tool result', async () => { const coordinator = new SessionContinuityCoordinator( HOST_EPOCH, diff --git a/packages/runtime-host/src/adapter/session-projector.ts b/packages/runtime-host/src/adapter/session-projector.ts index 944644e1c2..93f1f943e7 100644 --- a/packages/runtime-host/src/adapter/session-projector.ts +++ b/packages/runtime-host/src/adapter/session-projector.ts @@ -691,7 +691,9 @@ function projectSessionEvent( type: 'tool_result', ...base, contentOmitted: true, - isError: event.status === 'errored', + isError: event.status !== 'completed', + outcome: + event.status === 'interrupted' ? 'aborted' : event.status === 'errored' ? 'error' : 'success', content: { kind: 'text', text: '', diff --git a/packages/runtime-host/src/client/client-capability-channel.ts b/packages/runtime-host/src/client/client-capability-channel.ts index 4467f961ae..b805d4d6f3 100644 --- a/packages/runtime-host/src/client/client-capability-channel.ts +++ b/packages/runtime-host/src/client/client-capability-channel.ts @@ -304,6 +304,7 @@ export class ClientCapabilityChannel { this.#invocations.set(frame.invocationId, invocation); void this.#runInvocation(frame.invocationId, invocation, async (options) => decodeClientCapabilityResult({ + outcome: 'success', content: [], structuredContent: await registration.provider.callService!(frame, options), }), diff --git a/packages/runtime-host/src/protocol/client-capability.ts b/packages/runtime-host/src/protocol/client-capability.ts index 643c1b4e9c..ad46039f61 100644 --- a/packages/runtime-host/src/protocol/client-capability.ts +++ b/packages/runtime-host/src/protocol/client-capability.ts @@ -18,6 +18,7 @@ */ import { TOOL_ACTIVITY_KINDS, type ToolActivityKind } from '@maka/core/events'; +import { isToolCallOutcome, type ToolCallOutcome } from '@maka/core/tool-result-status'; import { decodeInteractionAnswer, projectInteractionFormRequest, @@ -74,6 +75,7 @@ export type ClientCapabilityContentBlock = | { readonly type: 'unknown'; readonly value: unknown }; export interface ClientCapabilityCallResult { + readonly outcome: ToolCallOutcome; readonly content: ClientCapabilityContentBlock[]; readonly structuredContent?: unknown; } @@ -649,12 +651,21 @@ export function decodeClientCapabilityHostFrame(value: unknown): ClientCapabilit export function decodeClientCapabilityResult(value: unknown): ClientCapabilityCallResult { const record = requireRecord(value, 'Client Capability result'); - assertOptionalExactKeys(record, 'Client Capability result', ['content'], ['structuredContent']); + assertOptionalExactKeys( + record, + 'Client Capability result', + ['outcome', 'content'], + ['structuredContent'], + ); + if (!isToolCallOutcome(record.outcome)) { + throw invalidProtocolFrame('Invalid Client Capability result outcome'); + } if (!Array.isArray(record.content) || record.content.length > 256) { throw invalidProtocolFrame('Invalid Client Capability result content'); } const content = record.content.map(decodeContentBlock); return { + outcome: record.outcome, content, ...(Object.hasOwn(record, 'structuredContent') ? { diff --git a/packages/runtime-host/src/protocol/index.ts b/packages/runtime-host/src/protocol/index.ts index b6dd7337e7..f8820feb46 100644 --- a/packages/runtime-host/src/protocol/index.ts +++ b/packages/runtime-host/src/protocol/index.ts @@ -101,7 +101,9 @@ export const RUNTIME_HOST_REGISTRATION_SCHEMA_VERSION = 1 as const; export const RUNTIME_HOST_PROTOCOL_VERSION = 0 as const; // Increment when the same protocol version no longer guarantees safe Client-Host // interoperability. Mismatches are rejected before domain commands are admitted. -export const RUNTIME_HOST_COMPATIBILITY_EPOCH = 161 as const; +export const RUNTIME_HOST_COMPATIBILITY_EPOCH = 162 as const; +// 162: Client Capability results require a tri-state outcome; session tool +// continuity also carries interrupted results. Older peers cannot decode it. // 161: Session transcript reads return the whole transcript under a byte budget, // and every page says whether it stops between two Turns. The windowed read's // range edges are gone, and the Turn landmark query takes a Turn to look up, so diff --git a/packages/runtime-host/src/protocol/session-continuity.ts b/packages/runtime-host/src/protocol/session-continuity.ts index 95bba3299f..33f81d0738 100644 --- a/packages/runtime-host/src/protocol/session-continuity.ts +++ b/packages/runtime-host/src/protocol/session-continuity.ts @@ -207,7 +207,7 @@ export type SessionToolEvent = | (SessionToolEventIdentity & { type: 'tool_result'; operationId?: string; - status: 'completed' | 'errored'; + status: 'completed' | 'errored' | 'interrupted'; sandboxFailureReason?: SandboxBoundaryFailureSignal['reason']; durationMs?: number; }) @@ -987,7 +987,11 @@ function decodeSessionToolEvent(value: unknown): SessionToolEvent { 'toolUseId', 'status', ]); - if (record.status !== 'completed' && record.status !== 'errored') { + if ( + record.status !== 'completed' && + record.status !== 'errored' && + record.status !== 'interrupted' + ) { throw invalidProtocolFrame('Invalid Session tool result status'); } if (record.status === 'completed' && record.sandboxFailureReason !== undefined) { diff --git a/packages/runtime-host/src/server/client-capability-coordinator.ts b/packages/runtime-host/src/server/client-capability-coordinator.ts index b1ce6da249..04883d8fef 100644 --- a/packages/runtime-host/src/server/client-capability-coordinator.ts +++ b/packages/runtime-host/src/server/client-capability-coordinator.ts @@ -26,6 +26,7 @@ import { type McpToolProvider, } from '@maka/runtime/mcp-tools'; import { type MakaTool } from '@maka/runtime/tool-runtime'; +import { requireToolCallOutcome } from '@maka/core/tool-result-status'; import type { RootExecutionDescriptor } from '@maka/core/runtime-invocation'; import { clientCapabilityScopeIdentity, @@ -649,6 +650,7 @@ export class HostClientCapabilityCoordinator implements ClientCapabilityService assertUniqueSnapshotToolIdentities(selected); const tools = [ ...buildMcpTools(this.#snapshotProvider(state?.initiatingProviderId, interactive), { + resultOutcome: (output) => requireToolCallOutcome(output.outcome), callTimeoutMs: DEFAULT_CALL_TIMEOUT_MS, categoryHint: 'client_capability', hostAdmission: 'client_capability', @@ -656,6 +658,7 @@ export class HostClientCapabilityCoordinator implements ClientCapabilityService executionLocation: 'remote', }), ...buildMcpTools(this.#snapshotProvider(state?.initiatingProviderId, trusted), { + resultOutcome: (output) => requireToolCallOutcome(output.outcome), callTimeoutMs: DEFAULT_CALL_TIMEOUT_MS, categoryHint: 'custom_tool', hostAdmission: 'client_capability', diff --git a/packages/runtime-host/src/server/session-continuity-coordinator.ts b/packages/runtime-host/src/server/session-continuity-coordinator.ts index 4e10ff14cf..95edb4d18d 100644 --- a/packages/runtime-host/src/server/session-continuity-coordinator.ts +++ b/packages/runtime-host/src/server/session-continuity-coordinator.ts @@ -2031,7 +2031,8 @@ function projectSessionEvent( type: event.type, ...identity, ...(event.operationId === undefined ? {} : { operationId: event.operationId }), - status: event.isError ? 'errored' : 'completed', + status: + event.outcome === 'aborted' ? 'interrupted' : event.isError ? 'errored' : 'completed', ...(event.isError && event.content.kind === 'text' && event.content.sandboxFailure ? { sandboxFailureReason: event.content.sandboxFailure.reason } : {}), diff --git a/packages/runtime/src/__tests__/computer-use-tools.test.ts b/packages/runtime/src/__tests__/computer-use-tools.test.ts index bb9d23be24..d011170413 100644 --- a/packages/runtime/src/__tests__/computer-use-tools.test.ts +++ b/packages/runtime/src/__tests__/computer-use-tools.test.ts @@ -1813,6 +1813,8 @@ describe('buildComputerUseTools — the `maka_computer` MakaTool', () => { const r = await callComputer(fakeBackend({ accessibility: false }), { action: 'wait' }); assert.match(r.text, /permission_missing/); assert.match(r.text, /Accessibility/); + assert.equal((r as { error?: string }).error, 'permission_missing'); + assert.equal((r as { outcome?: string }).outcome, 'error'); }); test('requests Accessibility once on first use while every action still preflights', async () => { @@ -2691,6 +2693,7 @@ describe('buildComputerUseTools — the `maka_computer` MakaTool', () => { const backend = fakeBackend(); const r = await callComputer(backend, { action: 'wait', duration: 0.001 }, ac.signal); assert.match(r.text, /aborted/); + assert.equal((r as { outcome?: string }).outcome, 'aborted'); assert.equal(backend.last, undefined, 'backend.run must not be called after abort'); }); diff --git a/packages/runtime/src/__tests__/computer-use-wait-for.test.ts b/packages/runtime/src/__tests__/computer-use-wait-for.test.ts index 97b7f8c75e..fc4979d1c4 100644 --- a/packages/runtime/src/__tests__/computer-use-wait-for.test.ts +++ b/packages/runtime/src/__tests__/computer-use-wait-for.test.ts @@ -75,7 +75,12 @@ async function waitFor( b: CuDispatchBackend, args: Record, observeFirst = true, -): Promise<{ text: string; modelText?: string; error?: string }> { +): Promise<{ + text: string; + modelText?: string; + error?: string; + outcome?: string; +}> { const [tool] = buildComputerUseTools({ backend: b }); const context = { abortSignal: new AbortController().signal, @@ -93,6 +98,7 @@ async function waitFor( text: string; modelText?: string; error?: string; + outcome?: string; }; } @@ -114,6 +120,7 @@ test('a wait for text to go returns when it goes', async () => { duration: 5, }); assert.match(result.text, /gone after/); + assert.equal(result.outcome, 'success'); }); test('a timeout hands back the window as it stands', async () => { @@ -122,6 +129,7 @@ test('a timeout hands back the window as it stands', async () => { duration: 0.6, }); assert.equal(result.error, 'timeout'); + assert.equal(result.outcome, 'error'); assert.match(result.text, /was still absent after/); // The whole question a model asks after a timeout is "what is there instead", // and making it spend another call on that is the round trip this removes. diff --git a/packages/runtime/src/__tests__/runtime-event-read-model.test.ts b/packages/runtime/src/__tests__/runtime-event-read-model.test.ts index e5b1dac4fb..2f93899f1c 100644 --- a/packages/runtime/src/__tests__/runtime-event-read-model.test.ts +++ b/packages/runtime/src/__tests__/runtime-event-read-model.test.ts @@ -293,6 +293,36 @@ function equivalentLegacyMessages(): StoredMessage[] { } describe('projectRuntimeEventsToStoredMessages', () => { + test('preserves declared aborts while projecting old result events unchanged', () => { + const oldEvents = baseEvents(); + const oldResult = projectRuntimeEventsToStoredMessages(oldEvents, { + invocations: [invocation], + }).messages.find((message) => message.type === 'tool_result'); + assert.equal(oldResult?.type === 'tool_result' && oldResult.outcome, undefined); + + const events = oldEvents.map( + (event): RuntimeEvent => + event.id === 'evt-tool-result' + ? { + ...event, + content: { + kind: 'function_response', + id: 'tool-1', + name: 'Read', + result: { kind: 'text', text: 'file contents' }, + isError: true, + outcome: 'aborted', + }, + } + : event, + ); + const result = projectRuntimeEventsToStoredMessages(events, { + invocations: [invocation], + }).messages.find((message) => message.type === 'tool_result'); + assert.equal(result?.type === 'tool_result' && result.isError, true); + assert.equal(result?.type === 'tool_result' && result.outcome, 'aborted'); + }); + test('streaming projection is equivalent to the batch read model', () => { const events = baseEvents(); const streamed = createRuntimeEventStoredMessageProjector({ invocations: [invocation] }); diff --git a/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts b/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts index aefc6d2551..99d59301a7 100644 --- a/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts +++ b/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts @@ -33,6 +33,53 @@ import type { import { ToolRuntime, type MakaTool } from '../tool-runtime.js'; describe('ToolRuntime durable boundary', () => { + it('commits declared outcomes once and leaves ordinary business status alone', async () => { + const outcomes: ToolOutcomeCommit[] = []; + const telemetry: string[] = []; + const harness = () => + makeHarness( + { + commitToolPrepared: async () => ({ + created: true, + runtimeEventSeq: 1, + }), + commitToolOutcome: async (input) => { + outcomes.push(input); + return { created: true, runtimeEventSeq: 2 }; + }, + }, + undefined, + 'run-1', + { + recordToolInvocation: (record) => telemetry.push(record.status), + }, + ); + for (const outcome of ['error', 'aborted', 'success'] as const) { + const attempt = harness(); + const target = tool(() => ({ + outcome, + content: [{ type: 'text', text: 'report' }], + })); + target.resultOutcome = (output) => (output as { outcome: typeof outcome }).outcome; + await attempt.execute(target); + const response = outcomes.at(-1)?.runtimeEvent.content; + assert.equal(response?.kind === 'function_response' && response.outcome, outcome); + assert.equal( + response?.kind === 'function_response' && response.isError === true, + outcome !== 'success', + ); + const live = attempt.events.at(-1); + assert.equal(live?.type === 'tool_result' && live.outcome, outcome); + assert.equal(live?.type === 'tool_result' && live.isError, outcome !== 'success'); + } + assert.deepEqual(telemetry, ['error', 'aborted', 'success']); + const ordinary = tool(() => ({ outcome: 'error', status: 'failed' })); + const ordinaryAttempt = harness(); + await ordinaryAttempt.execute(ordinary); + const ordinaryEvent = ordinaryAttempt.events.at(-1); + assert.equal(ordinaryEvent?.type === 'tool_result' && ordinaryEvent.isError, false); + }); + it('does not invoke the tool or publish a result when T1 fails', async () => { let implementationCalls = 0; const harness = makeHarness({ diff --git a/packages/runtime/src/computer-use-tools.ts b/packages/runtime/src/computer-use-tools.ts index 84c6ac5631..69f907ef1c 100644 --- a/packages/runtime/src/computer-use-tools.ts +++ b/packages/runtime/src/computer-use-tools.ts @@ -322,6 +322,21 @@ interface ComputerToolResult { includeScreenshotInModelOutput?: boolean; } +type SettledComputerToolResult = ComputerToolResult & + ( + | { outcome: 'success'; error?: never } + | { outcome: 'error'; error: ComputerUseErrorCode } + | { outcome: 'aborted'; error: 'user_stopped' } + ); + +function settleComputerResult(result: ComputerToolResult): SettledComputerToolResult { + if (result.error === 'user_stopped') + return { ...result, error: 'user_stopped', outcome: 'aborted' }; + if (result.error) return { ...result, error: result.error, outcome: 'error' }; + const { error: _error, ...content } = result; + return { ...content, outcome: 'success' }; +} + export const COMPUTER_USE_MODEL_SCREENSHOT_POLICY = { list_apps: 'never', launch_app: 'never', @@ -1460,7 +1475,7 @@ export function buildComputerUseTools(deps: { } } - const tool: MakaTool = { + const rawTool: MakaTool = { name: 'maka_computer', displayName: 'Maka Computer', // The kind every other builtin declares, and the reason `'computer'` is on @@ -1579,7 +1594,8 @@ export function buildComputerUseTools(deps: { args, { abortSignal, sessionId, turnId, toolCallId, emitProgress }, ): Promise => { - if (abortSignal.aborted) return { text: 'computer aborted before start' }; + if (abortSignal.aborted) + return { text: 'computer aborted before start', error: 'user_stopped' }; const input = snapshotComputerParams(computerParams.parse(args)); const includeScreenshotInModelOutput = shouldSendScreenshotToModel(input); // Before anything is claimed against a frame or dispatched: an argument @@ -1650,6 +1666,7 @@ export function buildComputerUseTools(deps: { } } return { + error: 'permission_missing', text: 'maka_computer failed: permission_missing — Accessibility not granted (System Settings → Privacy & Security → Accessibility)', }; } @@ -1657,6 +1674,7 @@ export function buildComputerUseTools(deps: { if (input.action === 'element_sequence') { if (!deps.backend.runSemantic || !deps.backend.captureObservation) { return { + error: 'unsupported_action', text: 'maka_computer.element_sequence failed: unsupported_action — ' + `${MISSING_CAPABILITY} Send the steps one at a time with click_element or ` + @@ -1665,6 +1683,7 @@ export function buildComputerUseTools(deps: { } if (!tcc.screenRecording) { return { + error: 'permission_missing', text: 'maka_computer.element_sequence failed: permission_missing — Screen Recording not granted (System Settings → Privacy & Security → Screen Recording)', }; } @@ -1869,7 +1888,11 @@ export function buildComputerUseTools(deps: { return { text: `${headline}${persistedTail}`, modelText: `${headline}\n${stepLines}${modelTail}`, - ...(stopped && isComputerUseErrorCode(stopped) ? { error: stopped } : {}), + ...(stopped + ? { + error: isComputerUseErrorCode(stopped) ? stopped : 'outcome_unknown', + } + : {}), ...(final?.screenshot ? { screenshot: { @@ -1883,6 +1906,7 @@ export function buildComputerUseTools(deps: { if (input.action === 'launch_app') { if (!deps.backend.launchApp) { return { + error: 'unsupported_action', text: 'maka_computer.launch_app failed: unsupported_action — ' + `${MISSING_CAPABILITY} Ask the user to open the application, then call ` + @@ -1913,6 +1937,7 @@ export function buildComputerUseTools(deps: { if (input.action === 'list_apps') { if (!deps.backend.listApps) { return { + error: 'unsupported_action', text: 'maka_computer.list_apps failed: unsupported_action — ' + `${MISSING_CAPABILITY} Name the application directly in action:"observe" ` + @@ -2014,6 +2039,7 @@ export function buildComputerUseTools(deps: { const record = observations.get(sessionId); if (!deps.backend.observeApp || !record?.appId) { return { + error: 'no_active_frame', text: 'maka_computer.wait failed: no_active_frame — a condition is checked against the window you last observed, and there is none yet. Observe first, or wait with only a duration.', }; } @@ -2042,6 +2068,7 @@ export function buildComputerUseTools(deps: { }; } return { + error: 'target_missing', text: 'maka_computer.wait failed: target_missing — the window being waited on is no longer there', }; } @@ -2084,6 +2111,7 @@ export function buildComputerUseTools(deps: { if (input.action === 'observe') { if (!deps.backend.observeApp) { return { + error: 'unsupported_action', text: 'maka_computer.observe failed: unsupported_action — ' + `${MISSING_CAPABILITY} Nothing on this computer can be read or driven; ` + @@ -2112,6 +2140,7 @@ export function buildComputerUseTools(deps: { const includeScreenshot = input.include_screenshot ?? false; if (includeScreenshot && !tcc.screenRecording) { return { + error: 'permission_missing', text: 'maka_computer.observe failed: permission_missing — Screen Recording not ' + 'granted (System Settings → Privacy & Security → Screen Recording). ' + @@ -2146,6 +2175,7 @@ export function buildComputerUseTools(deps: { const resolvedApp = await resolveAppName(input.app, abortSignal); if (resolvedApp && 'ambiguous' in resolvedApp) { return { + error: 'ambiguous_target', text: `maka_computer.observe failed: ambiguous_target — "${input.app}" matches ` + `${resolvedApp.ambiguous.join(', ')}. Name one of them.`, @@ -2271,6 +2301,7 @@ export function buildComputerUseTools(deps: { if (input.action === 'screenshot') { if (!deps.backend.observeApp) { return { + error: 'unsupported_action', text: 'maka_computer.screenshot failed: unsupported_action — ' + `${MISSING_CAPABILITY} Use action:"observe", which returns the same window ` + @@ -2279,6 +2310,7 @@ export function buildComputerUseTools(deps: { } if (!tcc.screenRecording) { return { + error: 'permission_missing', text: 'maka_computer.screenshot failed: permission_missing — ' + 'Screen Recording not granted ' + @@ -2305,7 +2337,10 @@ export function buildComputerUseTools(deps: { ); } if (!screenshotObservation.screenshot) { - return { text: 'maka_computer.screenshot failed: capture_failed' }; + return { + text: 'maka_computer.screenshot failed: capture_failed', + error: 'capture_failed', + }; } return { text: JSON.stringify({ @@ -2341,6 +2376,7 @@ export function buildComputerUseTools(deps: { ) { if (!deps.backend.runSemantic) { return { + error: 'unsupported_action', text: `maka_computer.${input.action} failed: unsupported_action — ` + `${MISSING_CAPABILITY} No element offers it either; report the limit instead ` + @@ -2349,6 +2385,7 @@ export function buildComputerUseTools(deps: { } if (!tcc.screenRecording) { return { + error: 'permission_missing', text: `maka_computer.${input.action} failed: permission_missing — Screen Recording not granted (System Settings → Privacy & Security → Screen Recording)`, }; } @@ -2619,6 +2656,7 @@ export function buildComputerUseTools(deps: { if (requiresActionLease) { if (!tcc.screenRecording) { return { + error: 'permission_missing', text: `maka_computer.${action.type} failed: permission_missing — Screen Recording not granted (System Settings → Privacy & Security → Screen Recording)`, }; } @@ -2631,6 +2669,7 @@ export function buildComputerUseTools(deps: { const capturing = action.type === 'screenshot'; if (capturing && !tcc.screenRecording) { return { + error: 'permission_missing', text: 'maka_computer failed: permission_missing — Screen Recording not granted (System Settings → Privacy & Security → Screen Recording)', }; } @@ -2789,6 +2828,11 @@ export function buildComputerUseTools(deps: { }; }, }; + const tool: MakaTool = { + ...rawTool, + resultOutcome: (output) => (output as SettledComputerToolResult).outcome, + impl: async (args, context) => settleComputerResult(await rawTool.impl(args, context)), + }; const debug = deps.debug; if (debug) { const dispatch = tool.impl; diff --git a/packages/runtime/src/mcp-tools.ts b/packages/runtime/src/mcp-tools.ts index 06ab9ec4d8..d8d17db3c0 100644 --- a/packages/runtime/src/mcp-tools.ts +++ b/packages/runtime/src/mcp-tools.ts @@ -91,6 +91,10 @@ export interface McpToolInvocationContext { } export interface BuildMcpToolsOptions { + /** Only a trusted adapter with a validated result envelope installs this reader. */ + resultOutcome?: ( + output: McpCallResult, + ) => import('@maka/core/tool-result-status').ToolCallOutcome; callTimeoutMs?: number; categoryHint?: ToolCategory; hostAdmission?: MakaTool['hostAdmission']; @@ -144,6 +148,7 @@ export function buildMcpToolsWithIdentities( // The MCP server remains the sole authority for the complete JSON // Schema. Runtime only carries the declaration to the AI SDK. parameters: jsonSchema(inputSchema), + ...(options.resultOutcome ? { resultOutcome: options.resultOutcome } : {}), ...(provider.prepareTool ? { prepareExecution: async (args: unknown, context) => { diff --git a/packages/runtime/src/runtime-event-backfill.ts b/packages/runtime/src/runtime-event-backfill.ts index be32b644d6..059ea60b60 100644 --- a/packages/runtime/src/runtime-event-backfill.ts +++ b/packages/runtime/src/runtime-event-backfill.ts @@ -286,6 +286,7 @@ export function backfillRuntimeEventsFromStoredMessages( name: call?.toolName ?? '', result: message.content, isError: message.isError, + ...(message.outcome ? { outcome: message.outcome } : {}), ...(message.providerExecuted !== undefined ? { providerExecuted: message.providerExecuted } : {}), diff --git a/packages/runtime/src/runtime-event-read-model.ts b/packages/runtime/src/runtime-event-read-model.ts index 7eb64a8398..785825393c 100644 --- a/packages/runtime/src/runtime-event-read-model.ts +++ b/packages/runtime/src/runtime-event-read-model.ts @@ -1225,6 +1225,7 @@ function projectFunctionResponse( ts: event.ts, toolUseId, isError: event.content.isError === true, + ...(event.content.outcome ? { outcome: event.content.outcome } : {}), content: resultContent, ...(event.content.providerExecuted !== undefined ? { providerExecuted: event.content.providerExecuted } diff --git a/packages/runtime/src/session-event-runtime-mapper.ts b/packages/runtime/src/session-event-runtime-mapper.ts index 4a00ece1c1..0a187578be 100644 --- a/packages/runtime/src/session-event-runtime-mapper.ts +++ b/packages/runtime/src/session-event-runtime-mapper.ts @@ -363,6 +363,7 @@ function mapBackendSessionEvent( name, result: event.content, ...(event.isError ? { isError: true as const } : {}), + ...(event.outcome ? { outcome: event.outcome } : {}), ...(event.providerExecuted !== undefined ? { providerExecuted: event.providerExecuted } : {}), diff --git a/packages/runtime/src/tool-runtime.ts b/packages/runtime/src/tool-runtime.ts index e7dc056cd4..cee1ee1703 100644 --- a/packages/runtime/src/tool-runtime.ts +++ b/packages/runtime/src/tool-runtime.ts @@ -18,6 +18,7 @@ */ import { decodeCanonicalToolResultContent } from '@maka/core/tool-result-record-schema'; +import { requireToolCallOutcome, type ToolCallOutcome } from '@maka/core/tool-result-status'; import { projectAgentSwarmResult } from '@maka/core/agent-swarm'; import { projectToolActivityArgs } from '@maka/core/tool-activity-args'; import { resolveCollaborationPermissionMode } from '@maka/core/collaboration'; @@ -219,6 +220,8 @@ export interface MakaTool

{ * settlement instead of detaching, so late side effects cannot outlive `exec`. */ impl: (args: P, ctx: MakaToolContext) => Promise | R; + /** Explicit call status for tools with a declared result envelope. */ + resultOutcome?(output: R): ToolCallOutcome; /** Best-effort compensation after T2 rejects a result that already produced side effects. */ compensateDurableOutcomeCommitFailure?: (input: { readonly result: unknown; @@ -449,6 +452,7 @@ interface DurableToolAttempt { isError: boolean, modelProjection: DurableToolResultProjection, durationMs?: number, + outcome?: ToolCallOutcome, ): Promise<{ id: string; operationId: string; ts: number }>; } @@ -1036,6 +1040,7 @@ export class ToolRuntime { turnId: string; toolUseId: string; isError: boolean; + outcome?: ToolCallOutcome; content: ToolResultContent; modelProjection: DurableToolResultProjection; durationMs?: number; @@ -1047,6 +1052,7 @@ export class ToolRuntime { input.isError, input.modelProjection, input.durationMs, + input.outcome, ); input.queue.push({ type: 'tool_result', @@ -1056,6 +1062,7 @@ export class ToolRuntime { toolUseId: input.toolUseId, ...(durableOutcome ? { operationId: durableOutcome.operationId } : {}), isError: input.isError, + ...(input.outcome ? { outcome: input.outcome } : {}), content: input.content, modelProjection: input.modelProjection, ...(input.durationMs !== undefined ? { durationMs: input.durationMs } : {}), @@ -1632,27 +1639,31 @@ export class ToolRuntime { const content = coerceResultContent(result); const projected = this.projectToolResult(tool, turnId, toolUseId, executionArgs, result); const modelProjection = isPromiseLike(projected) ? await projected : projected; + const outcome = tool.resultOutcome + ? requireToolCallOutcome(tool.resultOutcome(result)) + : deriveToolResultStatus(content, result); return { result, content, - isError: deriveToolResultStatus(content, result) !== 'success', + outcome, durationMs: this.input.now() - startedAt, modelProjection, }; }; - const { result, content, isError, durationMs, modelProjection } = + const { result, content, outcome, durationMs, modelProjection } = await prepareOperationValue(); + const isError = outcome !== 'success'; output.flush(); - // Keep the full provider-facing terminal classification. `isError` is - // sufficient for the durable response envelope, but it intentionally - // collapses `aborted` into an error bit and therefore cannot drive live - // tool status, telemetry, or subagent lifecycle projection. - const toolResultStatus = deriveToolResultStatus(content, result); + // Keep the full provider-facing terminal classification. `isError` + // remains a compatibility bit, while explicit outcomes preserve aborts + // for durable replay, live status, telemetry, and subagent projection. + const toolResultStatus = outcome; await this.commitAndPublishToolResult({ queue, turnId, toolUseId, isError, + ...(tool.resultOutcome ? { outcome } : {}), content, modelProjection, durationMs, @@ -2012,6 +2023,7 @@ export class ToolRuntime { isError: boolean, modelProjection: DurableToolResultProjection, durationMs: number | undefined, + outcome: ToolCallOutcome | undefined, ts: number, ): RuntimeEvent => ({ id: `${operationId}_response`, @@ -2031,6 +2043,7 @@ export class ToolRuntime { name: input.tool.name, result, ...(isError ? { isError: true } : {}), + ...(outcome ? { outcome } : {}), modelProjection, }, refs: { @@ -2048,13 +2061,14 @@ export class ToolRuntime { let committedOutcome: { id: string; operationId: string; ts: number } | undefined; return { operationId, - commitOutcome: async (result, isError, modelProjection, durationMs) => { + commitOutcome: async (result, isError, modelProjection, durationMs, outcome) => { if (committedOutcome) return committedOutcome; const responseEvent = buildResponseEvent( result, isError, modelProjection, durationMs, + outcome, this.input.now(), ); try { @@ -3264,7 +3278,7 @@ function uncertainOutcomeSignalFromError(error: unknown): ToolUncertainOutcomeSi }; } -function coerceResultContent(raw: unknown): ToolResultContent { +export function coerceResultContent(raw: unknown): ToolResultContent { if (typeof raw === 'string') return { kind: 'text', text: raw }; if (raw && typeof raw === 'object') { const obj = raw as { kind?: string; text?: string }; @@ -3410,7 +3424,7 @@ function isBoundaryAuthorityAttempt(toolName: string, args: unknown): boolean { ); } -function deriveToolResultStatus( +export function deriveToolResultStatus( content: ToolResultContent, raw?: unknown, ): ToolInvocationRecord['status'] { @@ -3446,10 +3460,25 @@ function deriveToolResultStatus( content.operation.failed ) return 'error'; - // All other structured results are successful tool executions. That includes - // ShellRun observations: their embedded process status stays model-visible, - // but reading or returning the observation itself succeeded. - return 'success'; + // Observing a failed process is still a successful call. Business JSON has + // the same legacy behavior unless its tool declares an outcome reader. + switch (content.kind) { + case 'text': + case 'json': + case 'file_diff': + case 'file_write': + case 'image': + case 'summary': + case 'archived_tool_result': + case 'web_search': + case 'shell_run': + case 'rive_workflow': + return 'success'; + default: { + const exhaustive: never = content; + return exhaustive; + } + } } function summarizeToolResultForTelemetry( diff --git a/packages/storage/src/__tests__/sqlite-runtime-store.test.ts b/packages/storage/src/__tests__/sqlite-runtime-store.test.ts index e3d444428a..89c432d87f 100644 --- a/packages/storage/src/__tests__/sqlite-runtime-store.test.ts +++ b/packages/storage/src/__tests__/sqlite-runtime-store.test.ts @@ -737,6 +737,36 @@ describe('SqliteRuntimeStore', () => { }); }); + it('retains an aborted tool outcome after reopening the T2 ledger', async () => { + await withStore(async (store, dbPath) => { + await commitPrepared(store); + const response = functionResponseEvent({ + content: { + kind: 'function_response', + id: 'provider-call-1', + name: 'Read', + result: { kind: 'text', text: 'stopped' }, + isError: true, + outcome: 'aborted', + }, + }); + await store.commitToolOutcome({ + operationId: 'operation-1', + journalEventId: 'operation-1_outcome', + runtimeEvent: response, + committedAt: 20, + }); + store.close(); + const reopened = createSqliteRuntimeStore(dbPath); + try { + const events = await reopened.readRuntimeEvents('session-1', 'run-1'); + assert.deepEqual(events.at(-1), response); + } finally { + reopened.close(); + } + }); + }); + it('keeps projected T2 prepared when its atomic model projection is missing', async () => { await withStore(async (store) => { await commitPrepared(store, { resultProjectionVersion: 1 }); diff --git a/packages/ui/src/__tests__/live-turn-projection.test.ts b/packages/ui/src/__tests__/live-turn-projection.test.ts index 250f6adb22..62e610d1a0 100644 --- a/packages/ui/src/__tests__/live-turn-projection.test.ts +++ b/packages/ui/src/__tests__/live-turn-projection.test.ts @@ -383,6 +383,21 @@ describe('applyLiveTurnEvent', () => { assert.equal(projection.steps[0]?.tools[0]?.status, 'interrupted'); }); + it('uses an explicit abort when the live result content is omitted', () => { + const projection = applyLiveTurnEvent(undefined, { + type: 'tool_result', + id: 'event-aborted', + turnId: 'turn-1', + toolUseId: 'tool-1', + isError: true, + outcome: 'aborted', + contentOmitted: true, + content: { kind: 'text', text: '' }, + ts: 101, + }); + assert.equal(projection.steps[0]?.tools[0]?.status, 'interrupted'); + }); + it('moves an output-first tool into its real step without duplicating or regressing it', () => { const output = applyLiveTurnEvent(undefined, { diff --git a/packages/ui/src/live-turn-projection.ts b/packages/ui/src/live-turn-projection.ts index 9f0130dd53..6a0b07fac2 100644 --- a/packages/ui/src/live-turn-projection.ts +++ b/packages/ui/src/live-turn-projection.ts @@ -476,7 +476,7 @@ function projectLiveTurnEvent( const tool: ToolActivityItem = { ...base, ...projectToolActivityIdentity(event), - status: toolResultActivityStatus(event.isError, event.content), + status: toolResultActivityStatus(event.isError, event.content, event.outcome), result: event.contentOmitted ? base.result : event.content, ...(event.durationMs !== undefined ? { durationMs: event.durationMs } : {}), }; diff --git a/packages/ui/src/materialize.ts b/packages/ui/src/materialize.ts index 7659bf74cd..a03e5af9ce 100644 --- a/packages/ui/src/materialize.ts +++ b/packages/ui/src/materialize.ts @@ -276,7 +276,7 @@ export function materializeTools( function materializeToolResultStatus( result: Extract, ): ToolActivityItem["status"] { - return toolResultActivityStatus(result.isError, result.content); + return toolResultActivityStatus(result.isError, result.content, result.outcome); } /** From 3e3ca4ca4681dc63b9033976c475d9c14fb738d9 Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:16:47 +0900 Subject: [PATCH 02/17] ci: rerun checks From 58550e7630c6951b13e5da9958c7c3d78ab449b9 Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:50:37 +0900 Subject: [PATCH 03/17] ci: rerun checks From c84f5a155ff907e632eee37b6b8e423ea55d4fb2 Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Fri, 18 Sep 2026 12:15:51 +0900 Subject: [PATCH 04/17] test(runtime-host): make session catalog CAS race deterministic --- .../src/__tests__/session-catalog-two-client-uds.test.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/runtime-host/src/__tests__/session-catalog-two-client-uds.test.ts b/packages/runtime-host/src/__tests__/session-catalog-two-client-uds.test.ts index ad5efe401e..60e58ec304 100644 --- a/packages/runtime-host/src/__tests__/session-catalog-two-client-uds.test.ts +++ b/packages/runtime-host/src/__tests__/session-catalog-two-client-uds.test.ts @@ -301,7 +301,7 @@ test('two Clients share stable Session creation, CAS configuration, and catalog sessionId: created.id, expectedRevision: configurationRevision, patch: { - permissionMode: 'bypass', + permissionMode: 'ask', orchestrationMode: 'default', }, }), From 9c9553e6f534c129a09a58338e345c87fee0a198 Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:25:05 +0900 Subject: [PATCH 05/17] fix(runtime): retain aborted outcome for thrown tool stops Generated-by: OpenAI Codex --- .../tool-runtime-durable-boundary.test.ts | 37 +++++++++++++++++++ packages/runtime/src/tool-runtime.ts | 16 +++++--- 2 files changed, 48 insertions(+), 5 deletions(-) diff --git a/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts b/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts index 99d59301a7..6e3275a129 100644 --- a/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts +++ b/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts @@ -80,6 +80,43 @@ describe('ToolRuntime durable boundary', () => { assert.equal(ordinaryEvent?.type === 'tool_result' && ordinaryEvent.isError, false); }); + it('commits a thrown stop as aborted without changing ordinary thrown failures', async () => { + const outcomes: ToolOutcomeCommit[] = []; + const telemetry: string[] = []; + const harness = makeHarness( + { + commitToolPrepared: async () => ({ created: true, runtimeEventSeq: 1 }), + commitToolOutcome: async (input) => { + outcomes.push(input); + return { created: true, runtimeEventSeq: 2 }; + }, + }, + undefined, + 'run-1', + { recordToolInvocation: (record) => telemetry.push(record.status) }, + ); + const controller = new AbortController(); + await harness.execute( + tool(() => { + controller.abort(new Error('user stopped')); + throw controller.signal.reason; + }), + controller.signal, + ); + const aborted = outcomes[0]?.runtimeEvent.content; + assert.equal(aborted?.kind === 'function_response' && aborted.outcome, 'aborted'); + const live = harness.events.at(-1); + assert.equal(live?.type === 'tool_result' && live.outcome, 'aborted'); + await harness.execute( + tool(() => { + throw new Error('ordinary failure'); + }), + ); + const failed = outcomes[1]?.runtimeEvent.content; + assert.equal(failed?.kind === 'function_response' && failed.outcome, 'error'); + assert.deepEqual(telemetry, ['aborted', 'error']); + }); + it('does not invoke the tool or publish a result when T1 fails', async () => { let implementationCalls = 0; const harness = makeHarness({ diff --git a/packages/runtime/src/tool-runtime.ts b/packages/runtime/src/tool-runtime.ts index cee1ee1703..7d60d94585 100644 --- a/packages/runtime/src/tool-runtime.ts +++ b/packages/runtime/src/tool-runtime.ts @@ -994,6 +994,7 @@ export class ToolRuntime { uncertainOutcome?: ToolUncertainOutcomeSignal, activityIdentity: ToolActivityIdentity = {}, attempt?: DurableToolAttempt, + outcome?: ToolCallOutcome, ): Promise { const content: ToolResultContent = { kind: 'text', @@ -1020,6 +1021,7 @@ export class ToolRuntime { name: toolName, result: content, isError: true, + ...(outcome ? { outcome } : {}), }, this.input.sessionId, ) ?? DURABLE_TOOL_RESULT_PROJECTION_FAILURE; @@ -1028,6 +1030,7 @@ export class ToolRuntime { turnId, toolUseId, isError: true, + ...(outcome ? { outcome } : {}), content, modelProjection, activityIdentity, @@ -1663,7 +1666,7 @@ export class ToolRuntime { turnId, toolUseId, isError, - ...(tool.resultOutcome ? { outcome } : {}), + outcome, content, modelProjection, durationMs, @@ -1766,6 +1769,7 @@ export class ToolRuntime { ); } const uncertainOutcome = uncertainOutcomeSignalFromError(err); + const failureOutcome = ctx.abortSignal.aborted && !uncertainOutcome ? 'aborted' : 'error'; const errorClass = uncertainOutcome ? 'OutcomeUnknown' : classifyError(err); const terminalFailure = coerceTerminalFailure( tool, @@ -1803,6 +1807,7 @@ export class ToolRuntime { turnId, toolUseId, isError: true, + outcome: failureOutcome, content: terminalFailure.content, modelProjection, durationMs, @@ -1817,7 +1822,7 @@ export class ToolRuntime { providerId: this.input.connection.providerType, modelId: this.input.modelId, durationMs, - status: 'error', + status: failureOutcome, errorClass, argsSummary: tool.categoryHint === 'computer_use' @@ -1832,7 +1837,7 @@ export class ToolRuntime { toolUseId, toolName: tool.name, durationMs, - status: 'error', + status: failureOutcome, errorClass, ...(sandboxError ? { sandbox: sandboxError } : {}), }); @@ -1857,6 +1862,7 @@ export class ToolRuntime { uncertainOutcome, activityIdentity, durableAttempt, + failureOutcome, ); this.input.recordToolInvocation?.({ sessionId: this.input.sessionId, @@ -1866,7 +1872,7 @@ export class ToolRuntime { providerId: this.input.connection.providerType, modelId: this.input.modelId, durationMs: Math.max(0, this.input.now() - startedAt), - status: 'error', + status: failureOutcome, errorClass, argsSummary: tool.categoryHint === 'computer_use' @@ -1880,7 +1886,7 @@ export class ToolRuntime { toolUseId, toolName: tool.name, durationMs: Math.max(0, this.input.now() - startedAt), - status: 'error', + status: failureOutcome, errorClass, ...(sandboxError ? { sandbox: sandboxError } : {}), }); From cd9d6cec9a6e1f4da0e5cafd4e5d67096a8ff3fa Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Fri, 18 Sep 2026 16:19:04 +0900 Subject: [PATCH 06/17] ci: rerun checks Generated-by: OpenAI Codex From 5d9f74bfe9a686424dbfd087141971b99b2692dc Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Fri, 18 Sep 2026 22:04:48 +0900 Subject: [PATCH 07/17] fix(runtime-host): preserve shared transcript tool outcome --- .../runtime-host/src/__tests__/session-transcript-pager.test.ts | 2 ++ packages/runtime-host/src/server/shared-session-transcript.ts | 1 + 2 files changed, 3 insertions(+) diff --git a/packages/runtime-host/src/__tests__/session-transcript-pager.test.ts b/packages/runtime-host/src/__tests__/session-transcript-pager.test.ts index b46a0f03eb..8a5c0516fc 100644 --- a/packages/runtime-host/src/__tests__/session-transcript-pager.test.ts +++ b/packages/runtime-host/src/__tests__/session-transcript-pager.test.ts @@ -141,6 +141,7 @@ test('projects durable transcript records before sharing them', async () => { ts: 2, toolUseId: 'tool-1', isError: false, + outcome: 'aborted', content: { kind: 'text', text: 'visible result' }, modelVisibility: 'hidden', providerOutput: { replay: 'private' }, @@ -216,6 +217,7 @@ test('projects durable transcript records before sharing them', async () => { assert.equal('displayText' in sharedDurable[0]!, false); assert.equal(sharedDurable[0]?.steeringEventId, 'steering-event-1'); assert.equal('providerOutput' in sharedDurable[1]!, false); + assert.equal(sharedDurable[1]?.outcome, 'aborted'); assert.equal('providerOptions' in sharedDurable[2]!, false); assert.deepEqual(sharedDurable[2]!.thinking, { text: 'visible thought' }); const projectedState = projectSharedSessionTranscriptMessage( diff --git a/packages/runtime-host/src/server/shared-session-transcript.ts b/packages/runtime-host/src/server/shared-session-transcript.ts index 243512859f..ced768c7e6 100644 --- a/packages/runtime-host/src/server/shared-session-transcript.ts +++ b/packages/runtime-host/src/server/shared-session-transcript.ts @@ -108,6 +108,7 @@ export function projectSharedSessionTranscriptMessage( ts: message.ts, toolUseId: message.toolUseId, isError: message.isError, + ...(message.outcome === undefined ? {} : { outcome: message.outcome }), content: message.content, ...(message.durationMs === undefined ? {} : { durationMs: message.durationMs }), ...(message.origin === undefined ? {} : { origin: message.origin }), From b5dc546c82b3b9cdc8be5b099e984e747f72d871 Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Sun, 20 Sep 2026 15:38:20 +0900 Subject: [PATCH 08/17] fix(runtime): keep declared outcomes out of tool content --- .../__tests__/client-capability-uds.test.ts | 12 ++++++++++- .../tool-runtime-durable-boundary.test.ts | 4 ++++ packages/runtime/src/tool-runtime.ts | 21 +++++++++++++++---- packages/ui/src/model-picker-internals.tsx | 1 + 4 files changed, 33 insertions(+), 5 deletions(-) diff --git a/packages/runtime-host/src/__tests__/client-capability-uds.test.ts b/packages/runtime-host/src/__tests__/client-capability-uds.test.ts index 1c9c87eb09..c0bd938b08 100644 --- a/packages/runtime-host/src/__tests__/client-capability-uds.test.ts +++ b/packages/runtime-host/src/__tests__/client-capability-uds.test.ts @@ -170,7 +170,12 @@ test('unknown Client Capability loads, invokes, and rebinds after UDS reconnect' } await accept({ kind: 'none' }); return { - outcome: frame.arguments.prefix === 'failure' ? 'error' : 'success', + outcome: + frame.arguments.prefix === 'failure' + ? 'error' + : frame.arguments.prefix === 'aborted' + ? 'aborted' + : 'success', content: [ { type: 'text', @@ -249,6 +254,11 @@ test('unknown Client Capability loads, invokes, and rebinds after UDS reconnect' outcome: 'error', content: [{ type: 'text', text: `failure:${largeValue}` }], }); + const aborted = await tool.impl({ prefix: 'aborted' }, toolContext); + assert.deepEqual(aborted, { + outcome: 'aborted', + content: [{ type: 'text', text: `aborted:${largeValue}` }], + }); await client.status(); await assert.rejects( async () => rejectedTool.impl({}, toolContext), diff --git a/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts b/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts index 6e3275a129..01bd89f044 100644 --- a/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts +++ b/packages/runtime/src/__tests__/tool-runtime-durable-boundary.test.ts @@ -68,6 +68,10 @@ describe('ToolRuntime durable boundary', () => { response?.kind === 'function_response' && response.isError === true, outcome !== 'success', ); + assert.deepEqual(response?.kind === 'function_response' ? response.result : undefined, { + kind: 'json', + value: { content: [{ type: 'text', text: 'report' }] }, + }); const live = attempt.events.at(-1); assert.equal(live?.type === 'tool_result' && live.outcome, outcome); assert.equal(live?.type === 'tool_result' && live.isError, outcome !== 'success'); diff --git a/packages/runtime/src/tool-runtime.ts b/packages/runtime/src/tool-runtime.ts index 7d60d94585..5af1768070 100644 --- a/packages/runtime/src/tool-runtime.ts +++ b/packages/runtime/src/tool-runtime.ts @@ -1639,12 +1639,17 @@ export class ToolRuntime { ) { throw new ToolResultLimitError(); } - const content = coerceResultContent(result); + const declaredOutcome = tool.resultOutcome + ? requireToolCallOutcome(tool.resultOutcome(result)) + : undefined; + const content = coerceResultContent( + declaredOutcome === undefined + ? result + : stripDeclaredToolOutcome(result, declaredOutcome), + ); const projected = this.projectToolResult(tool, turnId, toolUseId, executionArgs, result); const modelProjection = isPromiseLike(projected) ? await projected : projected; - const outcome = tool.resultOutcome - ? requireToolCallOutcome(tool.resultOutcome(result)) - : deriveToolResultStatus(content, result); + const outcome = declaredOutcome ?? deriveToolResultStatus(content, result); return { result, content, @@ -3301,6 +3306,14 @@ export function coerceResultContent(raw: unknown): ToolResultContent { return { kind: 'text', text: String(raw ?? '') }; } +function stripDeclaredToolOutcome(raw: unknown, outcome: ToolCallOutcome): unknown { + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) return raw; + const record = raw as Record; + if (record.outcome !== outcome) return raw; + const { outcome: _outcome, ...content } = record; + return content; +} + function coerceTerminalFailure( tool: MakaTool, cwd: string, diff --git a/packages/ui/src/model-picker-internals.tsx b/packages/ui/src/model-picker-internals.tsx index 8a41863182..5e6acf1bb6 100644 --- a/packages/ui/src/model-picker-internals.tsx +++ b/packages/ui/src/model-picker-internals.tsx @@ -37,6 +37,7 @@ export interface ModelPickerLeadingOption { providerType?: ProviderType; disabled?: boolean; } + type ModelChoiceValueFn = (choice: ModelMenuGroup['choices'][number]) => string; export function providerMarkIcon( From 6047e49d3a821d401728a8b92f77173f0d26d34f Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:54:33 +0900 Subject: [PATCH 09/17] fix(runtime): address capability outcome review --- .../session-inspector-panel-model.test.ts | 21 +++++++ .../cli/src/__tests__/pi-transcript.test.ts | 54 ++++++++++++++++ packages/cli/src/pi-transcript.ts | 20 +++--- packages/core/src/session-trace.ts | 7 ++- packages/runtime-host/src/protocol/index.ts | 3 +- .../session-trace-projection.test.ts | 48 ++++++++++++++ .../runtime/src/session-trace-projection.ts | 7 ++- packages/ui/src/__tests__/materialize.test.ts | 63 +++++++++++++++++++ packages/ui/src/materialize.ts | 38 +++++------ 9 files changed, 229 insertions(+), 32 deletions(-) diff --git a/apps/desktop/src/main/__tests__/session-inspector-panel-model.test.ts b/apps/desktop/src/main/__tests__/session-inspector-panel-model.test.ts index 1b39a27bc0..153208b05b 100644 --- a/apps/desktop/src/main/__tests__/session-inspector-panel-model.test.ts +++ b/apps/desktop/src/main/__tests__/session-inspector-panel-model.test.ts @@ -194,6 +194,27 @@ test('derives per-turn cost only from priced model-call step totals', () => { } }); +test('renders an interrupted tool as neutral while preserving the turn abort reason', () => { + const trace = traceWithSteps([ + { + kind: 'tool', + id: 'tool-1', + turnId: 'turn-1', + runId: 'run-1', + startedAt: 1, + endedAt: 2, + durationMs: 1, + toolName: 'Read', + status: 'interrupted', + }, + ]); + trace.turns[0]!.failure = { code: 'turn_aborted' }; + + const turn = deriveInspectorPanelModel(trace).turns[0]; + assert.equal(turn?.failureCode, 'turn_aborted'); + assert.equal(turn?.steps[0]?.failed, false); +}); + test('shows one compact diagnostic line for a failed history-compaction call', () => { const trace: SessionTrace = { schemaVersion: SESSION_TRACE_SCHEMA_VERSION, diff --git a/packages/cli/src/__tests__/pi-transcript.test.ts b/packages/cli/src/__tests__/pi-transcript.test.ts index 1c5c85bdd7..29c04c7d90 100644 --- a/packages/cli/src/__tests__/pi-transcript.test.ts +++ b/packages/cli/src/__tests__/pi-transcript.test.ts @@ -1419,6 +1419,60 @@ describe('Maka Pi TUI transcript', () => { ); }); + test('scopes stored tool results to their turn when opaque call ids repeat', () => { + const state = createMakaPiTranscriptState(); + replaceTranscriptWithStoredMessages(state, [ + { + type: 'tool_call', + id: 'call-1', + turnId: 'turn-1', + ts: 1, + toolName: 'Read', + args: { path: 'first.txt' }, + }, + { + type: 'tool_result', + id: 'result-1', + turnId: 'turn-1', + ts: 2, + toolUseId: 'call-1', + isError: false, + outcome: 'success', + content: { kind: 'text', text: 'first result' }, + }, + { + type: 'tool_call', + id: 'call-1', + turnId: 'turn-2', + ts: 3, + toolName: 'Read', + args: { path: 'second.txt' }, + }, + { + type: 'tool_result', + id: 'result-2', + turnId: 'turn-2', + ts: 4, + toolUseId: 'call-1', + isError: true, + outcome: 'aborted', + content: { kind: 'text', text: 'second result' }, + }, + ] satisfies StoredMessage[]); + + const tools = state.entries.filter((entry): entry is MakaPiToolEntry => entry.kind === 'tool'); + assert.deepEqual( + tools.map((tool) => ({ + status: toolStatus(tool), + result: tool.result, + })), + [ + { status: 'done', result: { kind: 'text', text: 'first result' } }, + { status: 'aborted', result: { kind: 'text', text: 'second result' } }, + ], + ); + }); + test('explains a stored tool call whose turn ended without a result', () => { const state = createMakaPiTranscriptState(); replaceTranscriptWithStoredMessages(state, [ diff --git a/packages/cli/src/pi-transcript.ts b/packages/cli/src/pi-transcript.ts index 7d0b8c95fc..926c0a5eba 100644 --- a/packages/cli/src/pi-transcript.ts +++ b/packages/cli/src/pi-transcript.ts @@ -1067,14 +1067,16 @@ function storedMessagesToTranscriptEntries( messages: readonly StoredMessage[], ): MakaPiTranscriptEntry[] { const entries: MakaPiTranscriptEntry[] = []; - const resultsByToolUseId = new Map( - messages - .filter( - (message): message is Extract => - message.type === 'tool_result', - ) - .map((message) => [message.toolUseId, message]), - ); + const resultsByTurnId = new Map< + string, + Map> + >(); + for (const message of messages) { + if (message.type !== 'tool_result') continue; + const turnResults = resultsByTurnId.get(message.turnId); + if (turnResults) turnResults.set(message.toolUseId, message); + else resultsByTurnId.set(message.turnId, new Map([[message.toolUseId, message]])); + } const turnStatusById = new Map( deriveTurnRecords(messages).map((turn) => [turn.turnId, turn.status]), ); @@ -1112,7 +1114,7 @@ function storedMessagesToTranscriptEntries( entries.push( storedToolToTranscriptEntry( message, - resultsByToolUseId.get(message.id), + resultsByTurnId.get(message.turnId)?.get(message.id), turnStatusById.get(message.turnId), ), ); diff --git a/packages/core/src/session-trace.ts b/packages/core/src/session-trace.ts index 71a624d810..b2f6966cd0 100644 --- a/packages/core/src/session-trace.ts +++ b/packages/core/src/session-trace.ts @@ -154,7 +154,7 @@ export interface TraceToolStep { toolName: string; toolCallId?: string; operationId?: string; - status: 'completed' | 'failed' | 'in_flight'; + status: 'completed' | 'failed' | 'interrupted' | 'in_flight'; /** * What the dispatch declared it would be safe to do on resume. Present on * normal executions too — it is a policy, not evidence that anything was @@ -538,7 +538,10 @@ function isToolStep(value: Record): boolean { isOptionalNonnegativeNumber(value.endedAt) && isOptionalNonnegativeNumber(value.durationMs) && [value.toolCallId, value.operationId, value.recoveryPolicy].every(isOptionalString) && - (value.status === 'completed' || value.status === 'failed' || value.status === 'in_flight') && + (value.status === 'completed' || + value.status === 'failed' || + value.status === 'interrupted' || + value.status === 'in_flight') && (value.recovered === undefined || isToolRecovery(value.recovered)) ); } diff --git a/packages/runtime-host/src/protocol/index.ts b/packages/runtime-host/src/protocol/index.ts index 5ceb6caad5..5e4c151e41 100644 --- a/packages/runtime-host/src/protocol/index.ts +++ b/packages/runtime-host/src/protocol/index.ts @@ -103,7 +103,8 @@ export const RUNTIME_HOST_REGISTRATION_SCHEMA_VERSION = 1 as const; export const RUNTIME_HOST_PROTOCOL_VERSION = 0 as const; // Increment when the same protocol version no longer guarantees safe Client-Host // interoperability. Mismatches are rejected before domain commands are admitted. -export const RUNTIME_HOST_COMPATIBILITY_EPOCH = 184 as const; +export const RUNTIME_HOST_COMPATIBILITY_EPOCH = 185 as const; +// 185: Session trace tool steps distinguish interrupted outcomes from failures. // 184: Client Capability results require a tri-state outcome; session tool // continuity also carries interrupted results. Older peers cannot decode the // changed strict shapes. diff --git a/packages/runtime/src/__tests__/session-trace-projection.test.ts b/packages/runtime/src/__tests__/session-trace-projection.test.ts index 99985ede86..d0697f0b51 100644 --- a/packages/runtime/src/__tests__/session-trace-projection.test.ts +++ b/packages/runtime/src/__tests__/session-trace-projection.test.ts @@ -229,6 +229,54 @@ describe('session trace projection', () => { assert.equal(failure.message, 'turn ended after tool failure'); }); + test('attributes an aborted tool outcome to the aborted turn, not a tool failure', () => { + const trace = projectSessionTrace({ + sessionId: 'session-1', + runtimeEvents: [ + event({ + id: 'dispatch-1', + ts: 1_000, + actions: { + toolDispatch: { + protocol: 't1_after_preflight_v1', + operationId: 'op-1', + providerToolCallId: 'tool-call-1', + toolName: 'Bash', + canonicalArgsHash: 'hash', + recoveryMode: 'replay_safe', + }, + }, + }), + event({ + id: 'response-1', + ts: 1_200, + role: 'tool', + author: 'tool', + content: { + kind: 'function_response', + id: 'tool-call-1', + name: 'Bash', + result: 'stopped', + isError: true, + outcome: 'aborted', + }, + }), + event({ + id: 'aborted-1', + ts: 1_300, + status: 'aborted', + }), + ], + modelCallAttempts: [], + }); + + const tool = trace.turns[0]?.steps.find((step) => step.kind === 'tool'); + assert.equal(tool?.kind === 'tool' ? tool.status : undefined, 'interrupted'); + assert.equal(trace.turns[0]?.failure?.code, 'turn_aborted'); + assert.equal(trace.turns[0]?.failure?.attributedToStepId, undefined); + assert.equal(isSessionTrace(trace), true, 'the Host protocol accepts interrupted tool steps'); + }); + test('projects a generic-lane function call as an in-flight tool step', () => { const trace = projectSessionTrace({ sessionId: 'session-1', diff --git a/packages/runtime/src/session-trace-projection.ts b/packages/runtime/src/session-trace-projection.ts index 204f1c53a1..84cd415057 100644 --- a/packages/runtime/src/session-trace-projection.ts +++ b/packages/runtime/src/session-trace-projection.ts @@ -373,7 +373,12 @@ function projectEventSteps(events: readonly RuntimeEvent[]): TraceStep[] { if (settled) { settled.endedAt = event.ts; settled.durationMs = Math.max(0, event.ts - settled.startedAt); - settled.status = response.isError === true ? 'failed' : 'completed'; + settled.status = + response.outcome === 'aborted' + ? 'interrupted' + : response.isError === true + ? 'failed' + : 'completed'; } continue; } diff --git a/packages/ui/src/__tests__/materialize.test.ts b/packages/ui/src/__tests__/materialize.test.ts index ca35c9d60b..5473f4b388 100644 --- a/packages/ui/src/__tests__/materialize.test.ts +++ b/packages/ui/src/__tests__/materialize.test.ts @@ -333,6 +333,69 @@ test('retains persisted nested tool activity identity', () => { }); }); +test('scopes persisted tool results to their turn when opaque call ids repeat', () => { + const turns = materializeTurns([ + userMsg('turn-1', 1, 'first'), + { + type: 'tool_call', + id: 'call-1', + turnId: 'turn-1', + ts: 2, + toolName: 'Read', + args: { path: 'first.txt' }, + }, + { + type: 'tool_result', + id: 'result-1', + turnId: 'turn-1', + ts: 3, + toolUseId: 'call-1', + isError: false, + outcome: 'success', + content: { kind: 'text', text: 'first result' }, + }, + userMsg('turn-2', 4, 'second'), + { + type: 'tool_call', + id: 'call-1', + turnId: 'turn-2', + ts: 5, + toolName: 'Read', + args: { path: 'second.txt' }, + }, + { + type: 'tool_result', + id: 'result-2', + turnId: 'turn-2', + ts: 6, + toolUseId: 'call-1', + isError: true, + outcome: 'aborted', + content: { kind: 'text', text: 'second result' }, + }, + ] satisfies StoredMessage[], 'en'); + + assert.deepEqual( + turns.map((turn) => ({ + turnId: turn.turnId, + status: turn.tools[0]?.status, + result: turn.tools[0]?.result, + })), + [ + { + turnId: 'turn-1', + status: 'completed', + result: { kind: 'text', text: 'first result' }, + }, + { + turnId: 'turn-2', + status: 'interrupted', + result: { kind: 'text', text: 'second result' }, + }, + ], + ); +}); + function shellRunResult(revision: number) { return { kind: "shell_run" as const, diff --git a/packages/ui/src/materialize.ts b/packages/ui/src/materialize.ts index ccc8d26e3a..f1221b702e 100644 --- a/packages/ui/src/materialize.ts +++ b/packages/ui/src/materialize.ts @@ -192,18 +192,23 @@ function systemNoteLabel(kind: string, data: unknown, locale: UiLocale): string export function materializeTools( messages: readonly StoredMessage[], ): ToolActivityItem[] { - const results = new Map( - messages - .filter((message) => message.type === "tool_result") - .map((message) => [message.toolUseId, message]), - ); + const resultsByTurnId = new Map< + string, + Map> + >(); + for (const message of messages) { + if (message.type !== "tool_result") continue; + const turnResults = resultsByTurnId.get(message.turnId); + if (turnResults) turnResults.set(message.toolUseId, message); + else resultsByTurnId.set(message.turnId, new Map([[message.toolUseId, message]])); + } const turnStatusById = new Map( deriveTurnRecords(messages).map((turn) => [turn.turnId, turn.status]), ); return messages .filter((message) => message.type === "tool_call") .map((call) => { - const result = results.get(call.id); + const result = resultsByTurnId.get(call.turnId)?.get(call.id); return { toolUseId: call.id, toolName: call.toolName, @@ -794,16 +799,7 @@ export function materializeTurns( } } - // Second pass: build the canonical tool map. Live tools are applied - // separately by overlayLiveTurn so streaming deltas never force settled - // history to rematerialize. - const toolItemByUseId = new Map( - foldShellRunToolActivities(materializeTools(messages)).map((tool) => [ - tool.toolUseId, - tool, - ]), - ); - // Third pass: rebuild each turn's render timeline from its storage-ordered + // Second pass: rebuild each turn's render timeline from its storage-ordered // messages, interleaving a step's thinking/text with its paired tools. The // timeline is the turn's only tool authority; `tools` is flattened out of it // so the two can never disagree about which tools a turn holds (a tool_call @@ -811,10 +807,14 @@ export function materializeTurns( // reaches exactly one timeline). for (const turnId of order) { const turn = byId.get(turnId)!; - turn.timeline = buildTurnTimeline( - messagesByTurn.get(turnId) ?? [], - toolItemByUseId, + const turnMessages = messagesByTurn.get(turnId) ?? []; + const toolItemByUseId = new Map( + foldShellRunToolActivities(materializeTools(turnMessages)).map((tool) => [ + tool.toolUseId, + tool, + ]), ); + turn.timeline = buildTurnTimeline(turnMessages, toolItemByUseId); turn.tools = timelineTools(turn.timeline); } From df116a6dc8f53eae88f58072654a57879d87060e Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Thu, 24 Sep 2026 11:14:45 +0900 Subject: [PATCH 10/17] fix(ui): retain scoped shell run folding --- packages/ui/src/materialize.ts | 77 +++++++++++++++++++++++++--------- 1 file changed, 58 insertions(+), 19 deletions(-) diff --git a/packages/ui/src/materialize.ts b/packages/ui/src/materialize.ts index f1221b702e..d5bba986b1 100644 --- a/packages/ui/src/materialize.ts +++ b/packages/ui/src/materialize.ts @@ -799,6 +799,23 @@ export function materializeTurns( } } + const toolCalls = messages.filter( + (message): message is Extract => + message.type === "tool_call", + ); + const foldedTools = foldScopedShellRunToolActivities( + materializeTools(messages).map((item, index) => ({ + turnId: toolCalls[index]!.turnId, + item, + })), + ); + const toolItemsByTurnId = new Map>(); + for (const { turnId, item } of foldedTools) { + const turnTools = toolItemsByTurnId.get(turnId); + if (turnTools) turnTools.set(item.toolUseId, item); + else toolItemsByTurnId.set(turnId, new Map([[item.toolUseId, item]])); + } + // Second pass: rebuild each turn's render timeline from its storage-ordered // messages, interleaving a step's thinking/text with its paired tools. The // timeline is the turn's only tool authority; `tools` is flattened out of it @@ -808,13 +825,10 @@ export function materializeTurns( for (const turnId of order) { const turn = byId.get(turnId)!; const turnMessages = messagesByTurn.get(turnId) ?? []; - const toolItemByUseId = new Map( - foldShellRunToolActivities(materializeTools(turnMessages)).map((tool) => [ - tool.toolUseId, - tool, - ]), + turn.timeline = buildTurnTimeline( + turnMessages, + toolItemsByTurnId.get(turnId) ?? new Map(), ); - turn.timeline = buildTurnTimeline(turnMessages, toolItemByUseId); turn.tools = timelineTools(turn.timeline); } @@ -841,24 +855,30 @@ export function finalAssistantReplyText(turn: TurnViewModel): string { * there, leaving an orphan tool row and a parent that never took the child's * revision. */ -export function foldShellRunToolActivities( - items: readonly ToolActivityItem[], -): ToolActivityItem[] { +interface ScopedToolActivityItem { + turnId: string; + item: ToolActivityItem; +} + +function foldScopedShellRunToolActivities( + items: readonly ScopedToolActivityItem[], +): ScopedToolActivityItem[] { const ownedRefs = new Set(); - for (const item of items) { + for (const { item } of items) { if (item.toolName === "Bash" && item.result?.kind === "shell_run") ownedRefs.add(item.result.ref); } - const folded: ToolActivityItem[] = []; + const folded: ScopedToolActivityItem[] = []; const parentIndexByRef = new Map(); const childResultsByRef = new Map(); - for (const item of items) { + for (const scopedItem of items) { + const { item } = scopedItem; const result = item.result?.kind === "shell_run" ? item.result : undefined; if (!result || item.toolName === "Bash") { if (result) parentIndexByRef.set(result.ref, folded.length); - folded.push(item); + folded.push(scopedItem); continue; } if (ownedRefs.has(result.ref)) { @@ -868,14 +888,14 @@ export function foldShellRunToolActivities( if (item.toolName === "Read" || item.toolName === "StopBackgroundTask") continue; } - folded.push(item); + folded.push(scopedItem); } for (const [ref, results] of childResultsByRef) { const index = parentIndexByRef.get(ref)!; const parent = folded[index]!; let current = - parent.result?.kind === "shell_run" ? parent.result : undefined; + parent.item.result?.kind === "shell_run" ? parent.item.result : undefined; let changed = false; for (const result of results) { const merged = mergeShellRunStateWithDiagnostics( @@ -888,18 +908,37 @@ export function foldShellRunToolActivities( changed = true; } } - if (changed && current) folded[index] = { ...parent, result: current }; + if (changed && current) + folded[index] = { ...parent, item: { ...parent.item, result: current } }; } return folded; } +export function foldShellRunToolActivities( + items: readonly ToolActivityItem[], +): ToolActivityItem[] { + return foldScopedShellRunToolActivities( + items.map((item) => ({ turnId: "", item })), + ).map(({ item }) => item); +} + function foldShellRunTurns( turns: readonly TurnViewModel[], ): readonly TurnViewModel[] { - return projectTurnTools( - turns, - foldShellRunToolActivities(turns.flatMap((turn) => turn.tools)), + const folded = foldScopedShellRunToolActivities( + turns.flatMap((turn) => + turn.tools.map((item) => ({ turnId: turn.turnId, item })), + ), + ); + const toolsByTurnId = new Map(); + for (const { turnId, item } of folded) { + const turnTools = toolsByTurnId.get(turnId); + if (turnTools) turnTools.push(item); + else toolsByTurnId.set(turnId, [item]); + } + return turns.map( + (turn) => projectTurnTools([turn], toolsByTurnId.get(turn.turnId) ?? [])[0]!, ); } From 56ed84dc2764968a334a11c4b36c6a482a833ae6 Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Fri, 25 Sep 2026 11:23:49 +0900 Subject: [PATCH 11/17] fix(ui): preserve unique cross-turn tool results --- .../src/main/__tests__/workhub-runtime.test.ts | 2 +- packages/ui/src/materialize.ts | 17 ++++++++++++++++- 2 files changed, 17 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/main/__tests__/workhub-runtime.test.ts b/apps/desktop/src/main/__tests__/workhub-runtime.test.ts index 34a0613ef0..a7a2d42d77 100644 --- a/apps/desktop/src/main/__tests__/workhub-runtime.test.ts +++ b/apps/desktop/src/main/__tests__/workhub-runtime.test.ts @@ -231,7 +231,7 @@ test('stop and resume receipts survive the client capability JSON boundary', asy const result = await f.runtime.actTasks(scope, 'turn', 'action', disposition === 'stop_work' ? { operation: 'stop', targetSessionId: 'target' } : { operation: 'resume', targetSessionId: 'target', resumesActionId: 'previous' }); - assert.deepEqual(decodeClientCapabilityResult({ content: [], structuredContent: result }).structuredContent, result); + assert.deepEqual(decodeClientCapabilityResult({ outcome: 'success', content: [], structuredContent: result }).structuredContent, result); assert.equal(Object.hasOwn(result, 'executionEvidence'), false); } }); diff --git a/packages/ui/src/materialize.ts b/packages/ui/src/materialize.ts index 71d5ac2233..86028b4f94 100644 --- a/packages/ui/src/materialize.ts +++ b/packages/ui/src/materialize.ts @@ -186,12 +186,24 @@ function systemNoteLabel(kind: string, data: unknown, locale: UiLocale): string export function materializeTools( messages: readonly StoredMessage[], ): ToolActivityItem[] { + const callTurnIdsByUseId = new Map>(); + const resultsByUseId = new Map< + string, + Extract + >(); const resultsByTurnId = new Map< string, Map> >(); for (const message of messages) { + if (message.type === "tool_call") { + const turnIds = callTurnIdsByUseId.get(message.id); + if (turnIds) turnIds.add(message.turnId); + else callTurnIdsByUseId.set(message.id, new Set([message.turnId])); + continue; + } if (message.type !== "tool_result") continue; + resultsByUseId.set(message.toolUseId, message); const turnResults = resultsByTurnId.get(message.turnId); if (turnResults) turnResults.set(message.toolUseId, message); else resultsByTurnId.set(message.turnId, new Map([[message.toolUseId, message]])); @@ -202,7 +214,10 @@ export function materializeTools( return messages .filter((message) => message.type === "tool_call") .map((call) => { - const result = resultsByTurnId.get(call.turnId)?.get(call.id); + const result = resultsByTurnId.get(call.turnId)?.get(call.id) + ?? (callTurnIdsByUseId.get(call.id)?.size === 1 + ? resultsByUseId.get(call.id) + : undefined); return { toolUseId: call.id, toolName: call.toolName, From a9264cdc89005c3d3d4ce1ae3716587876b4e855 Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Fri, 25 Sep 2026 14:12:26 +0900 Subject: [PATCH 12/17] fix(cli): preserve unique cross-turn tool results --- packages/cli/src/__tests__/pi-transcript.test.ts | 5 +++-- packages/cli/src/pi-transcript.ts | 14 +++++++++++++- 2 files changed, 16 insertions(+), 3 deletions(-) diff --git a/packages/cli/src/__tests__/pi-transcript.test.ts b/packages/cli/src/__tests__/pi-transcript.test.ts index 868057bf0d..c67878d7bf 100644 --- a/packages/cli/src/__tests__/pi-transcript.test.ts +++ b/packages/cli/src/__tests__/pi-transcript.test.ts @@ -2032,7 +2032,7 @@ describe('Maka Pi TUI transcript', () => { assert.deepEqual(after.slice(0, viewportTop), before.slice(0, viewportTop)); }); - test('replays WriteStdin as a human-readable operation row while merging its PTY revision into Bash', () => { + test('replays a unique cross-turn WriteStdin result while merging its PTY revision into Bash', () => { const state = createMakaPiTranscriptState(); const ref = 'maka://runtime/background-tasks/pty-1'; const rawInput = 'echo hello\r'; @@ -2071,7 +2071,7 @@ describe('Maka Pi TUI transcript', () => { { type: 'tool_result', id: 'write-result', - turnId: 'turn-2', + turnId: 'turn-results', ts: 4, toolUseId: 'write-pty', isError: false, @@ -2107,6 +2107,7 @@ describe('Maka Pi TUI transcript', () => { }, size: { cols: 100, rows: 30 }, }); + assert.equal(toolStatus(tools[1]), 'done'); assert.equal(tools[1]?.durationMs, undefined); refreshRunningShellRunElapsed(state, 3_000); assert.equal(tools[1]?.durationMs, undefined); diff --git a/packages/cli/src/pi-transcript.ts b/packages/cli/src/pi-transcript.ts index b5e5936fc2..b1021362ff 100644 --- a/packages/cli/src/pi-transcript.ts +++ b/packages/cli/src/pi-transcript.ts @@ -1067,12 +1067,21 @@ function storedMessagesToTranscriptEntries( messages: readonly StoredMessage[], ): MakaPiTranscriptEntry[] { const entries: MakaPiTranscriptEntry[] = []; + const callTurnIdsByUseId = new Map>(); + const resultsByUseId = new Map>(); const resultsByTurnId = new Map< string, Map> >(); for (const message of messages) { + if (message.type === 'tool_call') { + const turnIds = callTurnIdsByUseId.get(message.id); + if (turnIds) turnIds.add(message.turnId); + else callTurnIdsByUseId.set(message.id, new Set([message.turnId])); + continue; + } if (message.type !== 'tool_result') continue; + resultsByUseId.set(message.toolUseId, message); const turnResults = resultsByTurnId.get(message.turnId); if (turnResults) turnResults.set(message.toolUseId, message); else resultsByTurnId.set(message.turnId, new Map([[message.toolUseId, message]])); @@ -1114,7 +1123,10 @@ function storedMessagesToTranscriptEntries( entries.push( storedToolToTranscriptEntry( message, - resultsByTurnId.get(message.turnId)?.get(message.id), + resultsByTurnId.get(message.turnId)?.get(message.id) ?? + (callTurnIdsByUseId.get(message.id)?.size === 1 + ? resultsByUseId.get(message.id) + : undefined), turnStatusById.get(message.turnId), ), ); From 506da7dd4cfbcb4844ee176752cc199a7cb0e0a6 Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Fri, 25 Sep 2026 14:17:01 +0900 Subject: [PATCH 13/17] ci: retry flaky runtime host release test From e19ec01b1a342398e7b09ae6b23d10cda961484e Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:34:39 +0900 Subject: [PATCH 14/17] test(cli): allow slow plan host operations --- packages/cli/src/__tests__/acp-goal-plan-child-process.test.ts | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/packages/cli/src/__tests__/acp-goal-plan-child-process.test.ts b/packages/cli/src/__tests__/acp-goal-plan-child-process.test.ts index 945d146f4c..4872cc8946 100644 --- a/packages/cli/src/__tests__/acp-goal-plan-child-process.test.ts +++ b/packages/cli/src/__tests__/acp-goal-plan-child-process.test.ts @@ -582,7 +582,7 @@ describe('ACP Goal/Plan real Host routes', () => { }); test('submits and executes a Plan through the model, ACP, and real Host', { - timeout: 30_000, + timeout: 60_000, }, async () => { let modelCalls = 0; const modelToolNames: string[][] = []; @@ -1011,6 +1011,7 @@ describe('ACP Goal/Plan real Host routes', () => { }, { startRuntimeHost: true, + timeoutMs: 30_000, model: { id: 'goal-plan-fixture', thinkingLevels: [], From 4c744a81b13b0c78bddea757b00b57a8d8629848 Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:47:14 +0900 Subject: [PATCH 15/17] fix(transcript): fence reused tool call results --- .../cli/src/__tests__/pi-transcript.test.ts | 28 +++++++++++++++++++ packages/cli/src/pi-transcript.ts | 26 +++++++---------- packages/ui/src/__tests__/materialize.test.ts | 26 +++++++++++++++++ packages/ui/src/materialize.ts | 28 ++++++++----------- 4 files changed, 75 insertions(+), 33 deletions(-) diff --git a/packages/cli/src/__tests__/pi-transcript.test.ts b/packages/cli/src/__tests__/pi-transcript.test.ts index c67878d7bf..b9179b4c41 100644 --- a/packages/cli/src/__tests__/pi-transcript.test.ts +++ b/packages/cli/src/__tests__/pi-transcript.test.ts @@ -1471,6 +1471,34 @@ describe('Maka Pi TUI transcript', () => { ); }); + test('does not attach a retained old result to a reused call id at a history boundary', () => { + const state = createMakaPiTranscriptState(); + replaceTranscriptWithStoredMessages(state, [ + { + type: 'tool_result', + id: 'old-result', + turnId: 'turn-old', + ts: 1, + toolUseId: 'provider-call-1', + isError: true, + outcome: 'error', + content: { kind: 'text', text: 'old failure' }, + }, + { + type: 'tool_call', + id: 'provider-call-1', + turnId: 'turn-new', + ts: 2, + toolName: 'Read', + args: { path: 'new.txt' }, + }, + ] satisfies StoredMessage[]); + + const [tool] = state.entries.filter((entry): entry is MakaPiToolEntry => entry.kind === 'tool'); + assert.equal(tool?.result, undefined); + assert.notEqual(toolStatus(tool), 'errored'); + }); + test('explains a stored tool call whose turn ended without a result', () => { const state = createMakaPiTranscriptState(); replaceTranscriptWithStoredMessages(state, [ diff --git a/packages/cli/src/pi-transcript.ts b/packages/cli/src/pi-transcript.ts index b1021362ff..6dc94b52df 100644 --- a/packages/cli/src/pi-transcript.ts +++ b/packages/cli/src/pi-transcript.ts @@ -1067,24 +1067,21 @@ function storedMessagesToTranscriptEntries( messages: readonly StoredMessage[], ): MakaPiTranscriptEntry[] { const entries: MakaPiTranscriptEntry[] = []; - const callTurnIdsByUseId = new Map>(); - const resultsByUseId = new Map>(); - const resultsByTurnId = new Map< - string, - Map> + const latestCallsByUseId = new Map>(); + const resultsByCall = new Map< + Extract, + Extract >(); + // Append order bounds an opaque ID's ownership: a result belongs to the latest + // preceding call, and a later call with that ID starts a new interval. for (const message of messages) { if (message.type === 'tool_call') { - const turnIds = callTurnIdsByUseId.get(message.id); - if (turnIds) turnIds.add(message.turnId); - else callTurnIdsByUseId.set(message.id, new Set([message.turnId])); + latestCallsByUseId.set(message.id, message); continue; } if (message.type !== 'tool_result') continue; - resultsByUseId.set(message.toolUseId, message); - const turnResults = resultsByTurnId.get(message.turnId); - if (turnResults) turnResults.set(message.toolUseId, message); - else resultsByTurnId.set(message.turnId, new Map([[message.toolUseId, message]])); + const call = latestCallsByUseId.get(message.toolUseId); + if (call) resultsByCall.set(call, message); } const turnStatusById = new Map( deriveTurnRecords(messages).map((turn) => [turn.turnId, turn.status]), @@ -1123,10 +1120,7 @@ function storedMessagesToTranscriptEntries( entries.push( storedToolToTranscriptEntry( message, - resultsByTurnId.get(message.turnId)?.get(message.id) ?? - (callTurnIdsByUseId.get(message.id)?.size === 1 - ? resultsByUseId.get(message.id) - : undefined), + resultsByCall.get(message), turnStatusById.get(message.turnId), ), ); diff --git a/packages/ui/src/__tests__/materialize.test.ts b/packages/ui/src/__tests__/materialize.test.ts index eb77df4439..21184ac226 100644 --- a/packages/ui/src/__tests__/materialize.test.ts +++ b/packages/ui/src/__tests__/materialize.test.ts @@ -411,6 +411,32 @@ test('scopes persisted tool results to their turn when opaque call ids repeat', ); }); +test('does not attach a retained old result to a reused call id at a history boundary', () => { + const [tool] = materializeTools([ + { + type: 'tool_result', + id: 'old-result', + turnId: 'turn-old', + ts: 1, + toolUseId: 'provider-call-1', + isError: true, + outcome: 'error', + content: { kind: 'text', text: 'old failure' }, + }, + { + type: 'tool_call', + id: 'provider-call-1', + turnId: 'turn-new', + ts: 2, + toolName: 'Read', + args: { path: 'new.txt' }, + }, + ] satisfies StoredMessage[]); + + assert.equal(tool?.result, undefined); + assert.notEqual(tool?.status, 'errored'); +}); + function shellRunResult(revision: number) { return { kind: "shell_run" as const, diff --git a/packages/ui/src/materialize.ts b/packages/ui/src/materialize.ts index e11406eaa8..0d59e37245 100644 --- a/packages/ui/src/materialize.ts +++ b/packages/ui/src/materialize.ts @@ -186,27 +186,24 @@ function systemNoteLabel(kind: string, data: unknown, locale: UiLocale): string export function materializeTools( messages: readonly StoredMessage[], ): ToolActivityItem[] { - const callTurnIdsByUseId = new Map>(); - const resultsByUseId = new Map< + const latestCallsByUseId = new Map< string, - Extract + Extract >(); - const resultsByTurnId = new Map< - string, - Map> + const resultsByCall = new Map< + Extract, + Extract >(); + // Append order bounds an opaque ID's ownership: a result belongs to the latest + // preceding call, and a later call with that ID starts a new interval. for (const message of messages) { if (message.type === "tool_call") { - const turnIds = callTurnIdsByUseId.get(message.id); - if (turnIds) turnIds.add(message.turnId); - else callTurnIdsByUseId.set(message.id, new Set([message.turnId])); + latestCallsByUseId.set(message.id, message); continue; } if (message.type !== "tool_result") continue; - resultsByUseId.set(message.toolUseId, message); - const turnResults = resultsByTurnId.get(message.turnId); - if (turnResults) turnResults.set(message.toolUseId, message); - else resultsByTurnId.set(message.turnId, new Map([[message.toolUseId, message]])); + const call = latestCallsByUseId.get(message.toolUseId); + if (call) resultsByCall.set(call, message); } const turnStatusById = new Map( deriveTurnRecords(messages).map((turn) => [turn.turnId, turn.status]), @@ -214,10 +211,7 @@ export function materializeTools( return messages .filter((message) => message.type === "tool_call") .map((call) => { - const result = resultsByTurnId.get(call.turnId)?.get(call.id) - ?? (callTurnIdsByUseId.get(call.id)?.size === 1 - ? resultsByUseId.get(call.id) - : undefined); + const result = resultsByCall.get(call); return { toolUseId: call.id, toolName: call.toolName, From 85d357cf9f3a4b64d615354049119a8df0acd5bb Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Mon, 28 Sep 2026 15:27:23 +0900 Subject: [PATCH 16/17] fix(runtime): classify pre-dispatch and cancelled outcomes Generated-by: OpenAI Codex --- .../src/__tests__/computer-use-tools.test.ts | 44 +++++++++++++++++++ .../__tests__/tool-runtime-settlement.test.ts | 36 +++++++++++++++ packages/runtime/src/computer-use-tools.ts | 15 ++++--- packages/runtime/src/tool-runtime.ts | 6 ++- 4 files changed, 94 insertions(+), 7 deletions(-) diff --git a/packages/runtime/src/__tests__/computer-use-tools.test.ts b/packages/runtime/src/__tests__/computer-use-tools.test.ts index 6a5e8ae69d..100a48631c 100644 --- a/packages/runtime/src/__tests__/computer-use-tools.test.ts +++ b/packages/runtime/src/__tests__/computer-use-tools.test.ts @@ -1318,6 +1318,50 @@ describe('buildComputerUseTools — the `maka_computer` MakaTool', () => { assert.match(result.text, /stopped at step 1 of 1: outcome_unknown/); }); + test('a sequence reports a retired pre-dispatch step as a duplicate action', async () => { + const backend = fakeBackend(); + backend.observeApp = async () => observation(); + backend.captureObservation = async () => observation(); + let dispatches = 0; + backend.runSemantic = async () => { + dispatches += 1; + return { + outcome: { + ok: false, + error: 'dispatch_refused', + message: 'the executor refused before dispatch', + evidence: { path: 'none' }, + }, + }; + }; + const [tool] = buildComputerUseTools({ backend }); + const observed = (await tool.impl({ action: 'observe', app: 'Fixture' } as never, ctx())) as { + text: string; + }; + const observationId = JSON.parse(observed.text).observation_id as string; + const refused = (await tool.impl( + { + action: 'click_element', + observation_id: observationId, + element_id: '5', + } as never, + ctx(undefined, { toolCallId: 'refused' }), + )) as { error?: string }; + assert.equal(refused.error, 'dispatch_refused'); + + const sequence = (await tool.impl( + { + action: 'element_sequence', + observation_id: observationId, + steps: [{ label: 'Continue' }], + } as never, + ctx(undefined, { toolCallId: 'sequence' }), + )) as { error?: string; text: string }; + assert.equal(sequence.error, 'duplicate_action'); + assert.match(sequence.text, /stopped at step 0 of 1: retired_action/); + assert.equal(dispatches, 1); + }); + test('a stopped sequence preserves a partially delivered outcome over frame confirmation failure', async () => { const backend = fakeBackend(); backend.observeApp = async () => observation(); diff --git a/packages/runtime/src/__tests__/tool-runtime-settlement.test.ts b/packages/runtime/src/__tests__/tool-runtime-settlement.test.ts index 9bbb6eb4d4..0633090b8c 100644 --- a/packages/runtime/src/__tests__/tool-runtime-settlement.test.ts +++ b/packages/runtime/src/__tests__/tool-runtime-settlement.test.ts @@ -394,6 +394,42 @@ describe('ToolRuntime settlement', () => { } }); + it('records a thrown Bash cancellation as an aborted tool outcome', async () => { + const runtime = makeRuntime(); + const events: SessionEvent[] = []; + const bash = buildManagedBashTool({ + runForegroundBash: async () => { + throw Object.assign(new Error('cancelled'), { + code: 130, + stdout: '', + stderr: 'cancelled', + }); + }, + runBackgroundBash: async () => { + throw new Error('not used'); + }, + }); + + await runtime.settleToolCall({ + tool: bash, + turnId: 'turn-1', + stepId: 'step-1', + toolCallId: 'call-cancelled', + input: { command: 'printf cancelled', boundary_intent: 'current' }, + abortSignal: new AbortController().signal, + eventSink: { + push: (event) => events.push(event), + pushAndWaitUntilConsumed: async (event) => { + events.push(event); + }, + }, + }); + + const result = events.find((event) => event.type === 'tool_result'); + assert.equal(result?.type, 'tool_result'); + assert.equal(result?.outcome, 'aborted'); + }); + it('preserves live provider error mapping', async () => { const runtime = makeRuntime(); const events: SessionEvent[] = []; diff --git a/packages/runtime/src/computer-use-tools.ts b/packages/runtime/src/computer-use-tools.ts index 4f0ea8ab20..180cf81913 100644 --- a/packages/runtime/src/computer-use-tools.ts +++ b/packages/runtime/src/computer-use-tools.ts @@ -966,15 +966,18 @@ export function buildComputerUseTools(deps: { | 'target_changed' | 'capture_failed'; + function publicBindingFailureCode(reason: string): ComputerUseErrorCode | undefined { + if (isComputerUseErrorCode(reason)) return reason; + if (reason === 'retired_action') return 'duplicate_action'; + if (reason === 'invalid_binding') return 'stale_frame'; + return undefined; + } + function bindingFailure(reason: BindingFailureReason, action?: string): ComputerToolResult { // `retired_action` is an internal distinction, not a twenty-ninth word for // the model: it is the same fact as `duplicate_action` with a different // recovery, and the recovery is the sentence, not the code. - const error: ComputerUseErrorCode = isComputerUseErrorCode(reason) - ? reason - : reason === 'retired_action' - ? 'duplicate_action' - : 'stale_frame'; + const error = publicBindingFailureCode(reason) ?? 'stale_frame'; const tool = action ? `maka_computer.${action}` : 'maka_computer'; return { text: `${tool} failed: ${error} — ${BINDING_FAILURE_RECOVERY[reason]}`, @@ -1986,7 +1989,7 @@ export function buildComputerUseTools(deps: { modelText: `${headline}\n${stepLines}${modelTail}`, ...(stopped ? { - error: isComputerUseErrorCode(stopped) ? stopped : 'outcome_unknown', + error: publicBindingFailureCode(stopped) ?? 'outcome_unknown', } : closingBlock ? { error: closingBlock } diff --git a/packages/runtime/src/tool-runtime.ts b/packages/runtime/src/tool-runtime.ts index 40bfaad462..0a94afe67e 100644 --- a/packages/runtime/src/tool-runtime.ts +++ b/packages/runtime/src/tool-runtime.ts @@ -1778,7 +1778,6 @@ export class ToolRuntime { ); } const uncertainOutcome = uncertainOutcomeSignalFromError(err); - const failureOutcome = ctx.abortSignal.aborted && !uncertainOutcome ? 'aborted' : 'error'; const errorClass = uncertainOutcome ? 'OutcomeUnknown' : classifyError(err); const terminalFailure = coerceTerminalFailure( tool, @@ -1786,6 +1785,11 @@ export class ToolRuntime { executionArgs, err, ); + const failureOutcome = + !uncertainOutcome && + (ctx.abortSignal.aborted || terminalFailure?.content.status === 'cancelled') + ? 'aborted' + : 'error'; if (terminalFailure) { if (terminalFailure.sandboxDenied) { const denialKey = sandboxDenialKey(tool.name, this.input.header.cwd, executionArgs); From 98a663ffee751d5989ed59e2614f812ee80b01bd Mon Sep 17 00:00:00 2001 From: testikun <320479488+testikun@users.noreply.github.com> Date: Tue, 29 Sep 2026 11:10:43 +0900 Subject: [PATCH 17/17] fix(runtime): keep sequence failures model-safe --- .../src/__tests__/computer-use-tools.test.ts | 9 ++++++++- packages/runtime/src/computer-use-tools.ts | 17 +++++++++++------ 2 files changed, 19 insertions(+), 7 deletions(-) diff --git a/packages/runtime/src/__tests__/computer-use-tools.test.ts b/packages/runtime/src/__tests__/computer-use-tools.test.ts index 100a48631c..b6d876e5e9 100644 --- a/packages/runtime/src/__tests__/computer-use-tools.test.ts +++ b/packages/runtime/src/__tests__/computer-use-tools.test.ts @@ -1358,7 +1358,14 @@ describe('buildComputerUseTools — the `maka_computer` MakaTool', () => { ctx(undefined, { toolCallId: 'sequence' }), )) as { error?: string; text: string }; assert.equal(sequence.error, 'duplicate_action'); - assert.match(sequence.text, /stopped at step 0 of 1: retired_action/); + const modelOutput = tool.toModelOutput?.({ + toolCallId: 'sequence', + input: {}, + output: sequence, + }); + assert.match(JSON.stringify(modelOutput), /stopped at step 0 of 1: duplicate_action/); + assert.match(JSON.stringify(modelOutput), /Address a different element/); + assert.doesNotMatch(JSON.stringify(modelOutput), /retired_action/); assert.equal(dispatches, 1); }); diff --git a/packages/runtime/src/computer-use-tools.ts b/packages/runtime/src/computer-use-tools.ts index 180cf81913..15e4ea3157 100644 --- a/packages/runtime/src/computer-use-tools.ts +++ b/packages/runtime/src/computer-use-tools.ts @@ -1969,8 +1969,15 @@ export function buildComputerUseTools(deps: { } catch { final = undefined; } - const headline = stopped - ? `maka_computer.element_sequence stopped at step ${done.length} of ${input.steps.length}: ${stopped}` + const stoppedError = stopped + ? (publicBindingFailureCode(stopped) ?? 'outcome_unknown') + : undefined; + const stoppedRecovery = + stopped && stopped in BINDING_FAILURE_RECOVERY + ? BINDING_FAILURE_RECOVERY[stopped as BindingFailureReason] + : undefined; + const headline = stoppedError + ? `maka_computer.element_sequence stopped at step ${done.length} of ${input.steps.length}: ${stoppedError}${stoppedRecovery ? ` — ${stoppedRecovery}` : ''}` : closingBlock ? `maka_computer.element_sequence failed after ${done.length} of ${input.steps.length} steps: ${closingBlock} — ${SESSION_BLOCK_RECOVERY[closingBlock]}` : `maka_computer.element_sequence ok (${done.length} of ${input.steps.length} steps)`; @@ -1987,10 +1994,8 @@ export function buildComputerUseTools(deps: { return { text: `${headline}${persistedTail}`, modelText: `${headline}\n${stepLines}${modelTail}`, - ...(stopped - ? { - error: publicBindingFailureCode(stopped) ?? 'outcome_unknown', - } + ...(stoppedError + ? { error: stoppedError } : closingBlock ? { error: closingBlock } : {}),