Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
35 commits
Select commit Hold shift + click to select a range
28ca7fe
fix(runtime): preserve client capability tool outcomes
testikun Sep 17, 2026
4a169dd
Merge upstream/main into client capability outcomes
testikun Sep 18, 2026
3e3ca4c
ci: rerun checks
testikun Sep 18, 2026
58550e7
ci: rerun checks
testikun Sep 18, 2026
c84f5a1
test(runtime-host): make session catalog CAS race deterministic
testikun Sep 18, 2026
9c9553e
fix(runtime): retain aborted outcome for thrown tool stops
testikun Sep 18, 2026
cd9d6ce
ci: rerun checks
testikun Sep 18, 2026
927a47e
Merge remote-tracking branch 'upstream/main' into codex/fix-5449-capa…
testikun Sep 18, 2026
5d9f74b
fix(runtime-host): preserve shared transcript tool outcome
testikun Sep 18, 2026
a27b54e
Merge remote-tracking branch 'upstream/main' into codex/fix-5449-capa…
testikun Sep 20, 2026
b5dc546
fix(runtime): keep declared outcomes out of tool content
testikun Sep 20, 2026
c98ae77
merge: resolve main protocol epoch conflict for capability outcomes
testikun Sep 20, 2026
2ac5b51
Merge remote-tracking branch 'upstream/main' into codex/pr-5458-blockers
testikun Sep 21, 2026
ca37c76
Merge upstream main into capability outcomes
testikun Sep 21, 2026
a7d3f5e
Merge current main into capability outcomes
testikun Sep 22, 2026
926479d
merge: preserve capability outcomes on current main
testikun Sep 23, 2026
cae0e1e
merge: reconcile capability outcome epoch with current main
testikun Sep 23, 2026
edd2f15
Merge remote-tracking branch 'upstream/main' into codex/repair-open-5458
testikun Sep 24, 2026
6047e49
fix(runtime): address capability outcome review
testikun Sep 24, 2026
df116a6
fix(ui): retain scoped shell run folding
testikun Sep 24, 2026
c9074d1
Merge remote-tracking branch 'upstream/main' into codex/repair-open-5458
testikun Sep 25, 2026
56ed84d
fix(ui): preserve unique cross-turn tool results
testikun Sep 25, 2026
a9264cd
fix(cli): preserve unique cross-turn tool results
testikun Sep 25, 2026
506da7d
ci: retry flaky runtime host release test
testikun Sep 25, 2026
2cc64d1
merge: refresh capability outcomes on main
testikun Sep 28, 2026
e19ec01
test(cli): allow slow plan host operations
testikun Sep 28, 2026
4c744a8
fix(transcript): fence reused tool call results
testikun Sep 28, 2026
d21039d
Merge remote-tracking branch 'upstream/main' into codex/repair-open-5458
testikun Sep 28, 2026
85d357c
fix(runtime): classify pre-dispatch and cancelled outcomes
testikun Sep 28, 2026
14b172d
Merge remote-tracking branch 'upstream/main' into codex/repair-open-5458
testikun Sep 28, 2026
e681b56
Merge remote-tracking branch 'upstream/main' into codex/repair-open-5458
testikun Sep 29, 2026
98a663f
fix(runtime): keep sequence failures model-safe
testikun Sep 29, 2026
65096ab
Merge remote-tracking branch 'upstream/main' into codex/repair-open-5458
testikun Sep 30, 2026
b14e35e
Merge remote-tracking branch 'upstream/main' into codex/repair-open-5458
testikun Sep 30, 2026
cc62c18
Merge remote-tracking branch 'upstream/main' into codex/repair-open-5458
testikun Sep 30, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions apps/desktop/src/main/__tests__/browser-tools.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -293,6 +293,7 @@ describe('browser tool execution', () => {
);
assert.equal(resolved, 2);
assert.deepEqual(result, {
outcome: 'success',
content: [
{
type: 'text',
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -49,7 +49,7 @@ for (const action of ['accept', 'decline', 'cancel'] as const) {
const answered = await host.answer(pending.requestId, action === 'accept' ? { action, values } : { action });
assert.equal(answered.ok, true, JSON.stringify(answered));
const settled = await result;
assert.deepEqual(settled.result, { content: [{ type: 'text', text: 'complete' }] });
assert.deepEqual(settled.result, { outcome: 'success', content: [{ type: 'text', text: 'complete' }] });
assert.equal(server.calls.length, 2);
assert.notEqual(server.calls[0]?.id, server.calls[1]?.id);
assert.deepEqual(server.calls[1]?.params.arguments, server.calls[0]?.params.arguments);
Expand Down Expand Up @@ -82,7 +82,7 @@ test('Desktop handles a second MCP question as a new canonical Host form', { tim
assert.notEqual(second.requestId, first.requestId);
assert.equal(server.calls.length, 2);
assert.equal((await host.answer(second.requestId, { action: 'accept', values })).ok, true);
assert.deepEqual((await result).result, { content: [{ type: 'text', text: 'complete' }] });
assert.deepEqual((await result).result, { outcome: 'success', content: [{ type: 'text', text: 'complete' }] });
assert.equal(server.calls.length, 3);
assert.deepEqual(server.calls[1]?.params.inputResponses, { form: { action: 'decline' } });
assert.deepEqual(server.calls[2]?.params.inputResponses, { form: { action: 'accept', content: values } });
Expand Down
2 changes: 2 additions & 0 deletions apps/desktop/src/main/__tests__/mcp-runtime-e2e.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -156,6 +156,7 @@ test('MCP tools stay bound to the connection generation that advertised them', a
},
),
{
outcome: 'success',
content: [{ type: 'text', text: 'annotated:desktop-capability' }],
},
);
Expand Down Expand Up @@ -185,6 +186,7 @@ test('MCP tools stay bound to the connection generation that advertised them', a
},
),
{
outcome: 'success',
content: [
{ type: 'text', text: 'desktop-capability' },
{ type: 'text', text: '{"structuredContent":{"echoed":"desktop-capability"}}' },
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -636,12 +636,14 @@ test('refreshes native capabilities with a new immutable provider snapshot', asy
};

assert.deepEqual(await host.invokeCapability(frame), {
outcome: 'success',
content: [{ type: 'text', text: 'old' }],
});
implementation = 'new';
await controls?.refreshClientCapabilities();
assert.equal(host.capabilityRegistrations, 2);
assert.deepEqual(await host.invokeCapability(frame), {
outcome: 'success',
content: [{ type: 'text', text: 'new' }],
});

Expand Down Expand Up @@ -820,7 +822,7 @@ test('isolates an invalid dynamic MCP tool without dropping the Host connection'
serverId: 'desktop_mcp',
toolName: 'healthy_mcp',
}),
{ content: [{ type: 'text', text: 'healthy' }] },
{ outcome: 'success', content: [{ type: 'text', text: 'healthy' }] },
);

await candidate.close();
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -272,7 +272,7 @@ test('forwards JSON Schema native capability arguments to the MCP authority', as
arguments: {},
}),
),
{ content: [{ type: 'text', text: 'server result' }] },
{ outcome: 'success', content: [{ type: 'text', text: 'server result' }] },
);
assert.deepEqual(receivedArguments, {});
});
Expand Down Expand Up @@ -670,7 +670,7 @@ test('validates before admission and invokes the exact offered tool with Host co
admitted = true;
},
);
assert.deepEqual(result, { content: [{ type: 'text', text: 'Loaded' }] });
assert.deepEqual(result, { outcome: 'success', content: [{ type: 'text', text: 'Loaded' }] });
assert.deepEqual(received, {
args: { url: 'https://example.com/path' },
context: {
Expand Down Expand Up @@ -809,6 +809,7 @@ test('projects Computer Use screenshots and releases all native resources for a

const completed = await call(provider, computerFrame({ sessionId: 'completed-session', arguments: {} }));
assert.deepEqual(completed, {
outcome: 'success',
content: [
{ type: 'text', text: 'captured' },
{ type: 'image', data: 'aW1hZ2U=', mimeType: 'image/png' },
Expand Down Expand Up @@ -838,6 +839,19 @@ test('projects Computer Use screenshots and releases all native resources for a
await assert.rejects(() => call(provider, capabilityFrame()), /provider is closed/u);
});

test('projects native Computer Use refusal separately from its model text', async () => {
const backend = computerBackend();
backend.preflight = async () => ({ accessibility: false, screenRecording: true });
const provider = createDesktopNativeCapabilityProvider({
browserTools: [], resolveBrowserUrl: () => 'https://example.com/', releaseBrowserSession() {},
computerUseTools: buildComputerUseTools({ backend }), releaseDesktopInteractionSession() {},
});
const result = await call(provider, computerFrame({ arguments: { action: 'wait', duration: 0.001 } }));
assert.equal(result.outcome, 'error');
assert.match(result.content[0]?.type === 'text' ? result.content[0].text : '', /permission_missing/);
await provider.close();
});

test('does not advertise unavailable capability groups or dispatch unknown identities', async () => {
const provider = createDesktopNativeCapabilityProvider({
browserTools: [tool('browser_snapshot', z.object({}), async () => 'ok')],
Expand Down Expand Up @@ -900,7 +914,7 @@ test('dispatches through the same immutable tool snapshot it advertised', async
toolName: 'old_tool',
}),
),
{ content: [{ type: 'text', text: 'old implementation' }] },
{ outcome: 'success', content: [{ type: 'text', text: 'old implementation' }] },
);
await assert.rejects(
() =>
Expand Down Expand Up @@ -965,7 +979,7 @@ test('chunks a dynamic capability group beyond the single-offer tool limit', asy
arguments: {},
}),
),
{ content: [{ type: 'text', text: 'tool-64' }] },
{ outcome: 'success', content: [{ type: 'text', text: 'tool-64' }] },
);
await provider.close();
});
Expand Down Expand Up @@ -1140,7 +1154,7 @@ test('publishes identified tools under their real normalized MCP identity', asyn
arguments: {},
}),
),
{ content: [{ type: 'text', text: 'echo result' }] },
{ outcome: 'success', content: [{ type: 'text', text: 'echo result' }] },
);
assert.deepEqual(
await call(
Expand All @@ -1152,7 +1166,7 @@ test('publishes identified tools under their real normalized MCP identity', asyn
arguments: {},
}),
),
{ content: [{ type: 'text', text: 'run result' }] },
{ outcome: 'success', content: [{ type: 'text', text: 'run result' }] },
);
await provider.close();
});
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -194,6 +194,27 @@ test('derives per-turn cost only from priced model-call step totals', () => {
}
});

test('renders an interrupted tool as neutral while preserving the turn abort reason', () => {
const trace = traceWithSteps([
{
kind: 'tool',
id: 'tool-1',
turnId: 'turn-1',
runId: 'run-1',
startedAt: 1,
endedAt: 2,
durationMs: 1,
toolName: 'Read',
status: 'interrupted',
},
]);
trace.turns[0]!.failure = { code: 'turn_aborted' };

const turn = deriveInspectorPanelModel(trace).turns[0];
assert.equal(turn?.failureCode, 'turn_aborted');
assert.equal(turn?.steps[0]?.failed, false);
});

test('shows one compact diagnostic line for a failed history-compaction call', () => {
const trace: SessionTrace = {
schemaVersion: SESSION_TRACE_SCHEMA_VERSION,
Expand Down
2 changes: 1 addition & 1 deletion apps/desktop/src/main/__tests__/workhub-runtime.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -231,7 +231,7 @@ test('stop and resume receipts survive the client capability JSON boundary', asy
const result = await f.runtime.actTasks(scope, 'turn', 'action', disposition === 'stop_work'
? { operation: 'stop', targetSessionId: 'target' }
: { operation: 'resume', targetSessionId: 'target', resumesActionId: 'previous' });
assert.deepEqual(decodeClientCapabilityResult({ content: [], structuredContent: result }).structuredContent, result);
assert.deepEqual(decodeClientCapabilityResult({ outcome: 'success', content: [], structuredContent: result }).structuredContent, result);
assert.equal(Object.hasOwn(result, 'executionEvidence'), false);
}
});
5 changes: 5 additions & 0 deletions apps/desktop/src/main/computer-use-real-model-policy.ts
Original file line number Diff line number Diff line change
Expand Up @@ -109,12 +109,14 @@ export function applyComputerUseRealModelPolicy(
totalActions += 1;
if (totalActions > policy.maxTotalActions) {
return {
outcome: 'error',
text: 'maka_computer failed: total_action_budget_exceeded',
error: 'total_action_budget_exceeded',
};
}
if (!allowed.has(action)) {
return {
outcome: 'error',
text: `maka_computer.${action} failed: unsupported_action_policy`,
error: 'unsupported_action_policy',
};
Expand All @@ -125,6 +127,7 @@ export function applyComputerUseRealModelPolicy(
&& (typeof app !== 'string' || !allowedApps.has(app))
) {
return {
outcome: 'error',
text: `maka_computer.${action} failed: target_policy_mismatch`,
error: 'target_policy_mismatch',
};
Expand All @@ -142,6 +145,7 @@ export function applyComputerUseRealModelPolicy(
)
) {
return {
outcome: 'error',
text: `maka_computer.${action} failed: target_policy_mismatch`,
error: 'target_policy_mismatch',
};
Expand All @@ -150,6 +154,7 @@ export function applyComputerUseRealModelPolicy(
actionCounts.set(action, actionCount);
if (actionCount > (policy.maxActionCounts[action] ?? 0)) {
return {
outcome: 'error',
text: `maka_computer.${action} failed: action_budget_exceeded`,
error: 'action_budget_exceeded',
};
Expand Down
16 changes: 11 additions & 5 deletions apps/desktop/src/main/runtime-host-native-capabilities.ts
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,8 @@
import { Buffer } from "node:buffer";
import type { ComputerUseToolSet } from '@maka/runtime/computer-use-tools';
import type { MakaTool } from '@maka/runtime/tool-runtime';
import { coerceResultContent, deriveToolResultStatus } from '@maka/runtime/tool-runtime';
import { requireToolCallOutcome } from '@maka/core/tool-result-status';
import {
createOAuthPresentationClientProvider,
type ClientCapabilityProvider,
Expand Down Expand Up @@ -730,6 +732,9 @@ async function projectToolResult(
input: unknown,
output: unknown,
): Promise<ClientCapabilityCallResult> {
const outcome = tool.resultOutcome
? requireToolCallOutcome(tool.resultOutcome(output))
: deriveToolResultStatus(coerceResultContent(output), output);
const modelOutput = tool.toModelOutput
? await tool.toModelOutput({
toolCallId,
Expand All @@ -739,24 +744,25 @@ async function projectToolResult(
: undefined;
if (!modelOutput) {
return typeof output === "string"
? { content: [{ type: "text", text: output }] }
: { content: [], structuredContent: output };
? { outcome, content: [{ type: "text", text: output }] }
: { outcome, content: [], structuredContent: output };
}
switch (modelOutput.type) {
case "text":
case "error-text":
return { content: [{ type: "text", text: modelOutput.value }] };
return { outcome, content: [{ type: "text", text: modelOutput.value }] };
case "json":
case "error-json":
return { content: [], structuredContent: modelOutput.value };
return { outcome, content: [], structuredContent: modelOutput.value };
case "execution-denied":
return {
outcome,
content: [
{ type: "text", text: modelOutput.reason ?? "Execution denied" },
],
};
case "content":
return { content: modelOutput.value.map(projectContentPart) };
return { outcome, content: modelOutput.value.map(projectContentPart) };
}
}

Expand Down
26 changes: 26 additions & 0 deletions packages/cli/src/__tests__/acp-session-event-mapper.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -355,6 +355,32 @@ describe('ACP Session event mapper', () => {
assert.equal(notifications.length, count);
});

test('updates ACP host status when only the explicit outcome changes', async () => {
const notifications: SessionNotification[] = [];
const mapper = eventMapper(notifications);
for (const outcome of ['error', 'aborted'] as const) {
await mapper.accept(
event({
type: 'tool_result',
toolUseId: 'tool',
contentOmitted: true,
isError: true,
outcome,
content: { kind: 'text', text: '' },
}),
);
}
const updates = notifications.map(toolUpdate);
assert.deepEqual(
updates.map((update) => (update._meta?.maka as { hostStatus?: string })?.hostStatus),
['errored', 'interrupted'],
);
assert.deepEqual(
updates.map((update) => update.status),
['failed', 'failed'],
);
});

test('result before start creates one terminal card and late start only fills identity', async () => {
const notifications: SessionNotification[] = [];
const mapper = eventMapper(notifications);
Expand Down
10 changes: 8 additions & 2 deletions packages/cli/src/__tests__/mcp-form-host-integration.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -56,7 +56,10 @@ for (const action of ['accept', 'decline', 'cancel'] as const) {
);
assert.equal(answered.ok, true, JSON.stringify(answered));
const settled = await result;
assert.deepEqual(settled.result, { content: [{ type: 'text', text: 'complete' }] });
assert.deepEqual(settled.result, {
outcome: 'success',
content: [{ type: 'text', text: 'complete' }],
});
assert.equal(server.calls.length, 2);
assert.notEqual(server.calls[0]?.id, server.calls[1]?.id);
assert.deepEqual(server.calls[1]?.params.arguments, server.calls[0]?.params.arguments);
Expand Down Expand Up @@ -92,7 +95,10 @@ test('TUI handles a second MCP question as a new canonical Host form', {
assert.notEqual(second.requestId, first.requestId);
assert.equal(server.calls.length, 2);
assert.equal((await host.answer(second.requestId, { action: 'accept', values })).ok, true);
assert.deepEqual((await result).result, { content: [{ type: 'text', text: 'complete' }] });
assert.deepEqual((await result).result, {
outcome: 'success',
content: [{ type: 'text', text: 'complete' }],
});
assert.equal(server.calls.length, 3);
assert.deepEqual(server.calls[1]?.params.inputResponses, { form: { action: 'decline' } });
assert.deepEqual(server.calls[2]?.params.inputResponses, {
Expand Down
Loading