Skip to content

Commit f46c5cd

Browse files
ericallamTrigger.dev RepoOps
authored andcommitted
feat(dashboard-agent): move the agent to Claude Sonnet 5, with per-role model overrides and a managed summary prompt
feat(dashboard-agent): move the agent to Claude Sonnet 5, with per-role model overrides and a managed summary prompt The in-dashboard agent now answers with Claude Sonnet 5 for its main turns, the code and watch prompts, the warm first-turn step, compaction summaries, attention wakes and the turn eval judge. Claude Haiku 4.5 stays on chat titles. - Adds the Bedrock inference profile mapping for Sonnet 5 and keeps the Sonnet 4.6 entry so stored prompt versions that still name it resolve. - Switches thinking off for the bounded calls (summaries, attention wakes) so their output caps are all answer. - Main turns and the warm first-turn step pass each model's documented output ceiling explicitly, so an unknown-to-the-provider id no longer falls back to a 4096-token cap. - The compaction summariser is a managed prompt (`dashboard-agent-summary`), so its text and model can be versioned and overridden from the dashboard like the system prompt. - Each role's model can be overridden per environment with `DASHBOARD_AGENT_MODEL`, `DASHBOARD_AGENT_SUMMARY_MODEL`, `DASHBOARD_AGENT_JUDGE_MODEL` and `DASHBOARD_AGENT_TITLE_MODEL`, with the above as defaults. - No dependency changes; the installed providers accept the new id. Mono-RevId: c01a1daba486863c535fede891d96259fa6b7386
1 parent 5bcd1a9 commit f46c5cd

17 files changed

Lines changed: 364 additions & 77 deletions

‎apps/webapp/app/components/runs/v3/agent/AgentMessageView.tsx‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -132,8 +132,11 @@ export function renderPart(part: UIMessage["parts"][number], i: number) {
132132
return p.text ? <AssistantResponse key={i} text={p.text} headerLabel="" /> : null;
133133
}
134134

135-
// Reasoning — amber-bordered italic block
135+
// Reasoning — amber-bordered italic block. Models that think adaptively with the
136+
// text omitted (Sonnet 5 and later by default) still emit the part with no text;
137+
// an empty block would be a bare amber bar, so those render nothing.
136138
if (type === "reasoning") {
139+
if (!p.text) return null;
137140
return (
138141
<div key={i} className="border-l-2 border-amber-500/40 pl-2">
139142
<ChatBubble>

‎apps/webapp/app/env.server.ts‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -263,6 +263,11 @@ const EnvironmentSchema = z
263263
// uses its own key on the Trigger side. When unset, Head Start is disabled
264264
// and the first turn falls back to the normal cold-start path.
265265
ANTHROPIC_API_KEY: z.string().optional(),
266+
// The model the dashboard agent's head-start step runs on (canonical `claude-…`
267+
// id, default in the agent package). The internal seam reads process.env
268+
// directly; this entry documents it webapp-side. The agent run reads its own
269+
// DASHBOARD_AGENT_*_MODEL vars from the agent project's environment.
270+
DASHBOARD_AGENT_MODEL: z.string().optional(),
266271
// Selects the dashboard agent's LLM provider (default anthropic). The internal
267272
// seam reads process.env directly; this entry validates the value webapp-side.
268273
DASHBOARD_AGENT_MODEL_PROVIDER: z.preprocess(

‎apps/webapp/app/services/dashboardAgentHeadStart.server.ts‎

Lines changed: 5 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,10 +1,11 @@
1-
import { DASHBOARD_AGENT_MODEL } from "@internal/dashboard-agent/tool-schemas";
21
import { systemPromptFor, toolSchemasFor } from "@internal/dashboard-agent/prompt-assembly";
32
import {
43
describePromptPrefix,
54
promptCacheAttributes,
65
} from "@internal/dashboard-agent/prompt-prefix";
76
import {
7+
dashboardAgentModel,
8+
maxOutputTokensFor,
89
resolveDashboardAgentModel,
910
withCacheBreakpoint,
1011
} from "@internal/dashboard-agent/model-provider";
@@ -109,7 +110,9 @@ export async function startDashboardAgentHeadStart(params: {
109110
run: async ({ chat: helper }) =>
110111
streamText({
111112
...helper.toStreamTextOptions({ tools }),
112-
model: resolveDashboardAgentModel(DASHBOARD_AGENT_MODEL),
113+
model: resolveDashboardAgentModel(dashboardAgentModel(env.DASHBOARD_AGENT_MODEL)),
114+
// Same ceiling the agent run uses; the pinned provider caps an unknown id at 4096.
115+
maxOutputTokens: maxOutputTokensFor(dashboardAgentModel(env.DASHBOARD_AGENT_MODEL)),
113116
// A structured system message, not a bare string: without provider options
114117
// the provider neither writes nor reads the cache, so this call paid full price
115118
// for the prefix and the agent's step 2 then paid for a fresh write. The tool

‎internal-packages/dashboard-agent/src/agent-runtime.ts‎

Lines changed: 9 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -169,13 +169,20 @@ export function trackSeededInvestigation(
169169
* The action turn in flight, set by `onAction` when it returns `chat.turn()` and read
170170
* by the turn's hooks: a wake turn runs without tools (it reports what the check
171171
* established and carries no delegated token), a wake the narration plan marks
172-
* `haiku` runs on the small model with a bounded budget, and neither kind is judged
172+
* `bounded` runs with an output cap, and neither kind is judged
173173
* by the eval. `onTurnStart` drops a marker the previous turn left behind, so a
174174
* settlement that failed on both attempts never taints the next typed turn.
175175
*/
176+
/**
177+
* The canonical `"anthropic:<id>"` the current turn's model call resolved to, set in
178+
* `run()` and read by `onTurnComplete`, so the eval row records the model that
179+
* answered rather than the one the deployed prompt version names.
180+
*/
181+
export const turnModelKey = locals.create<string>("dashboard-agent.turnModel");
182+
176183
export const pendingActionTurnKey = locals.create<{
177184
kind: "wake" | "investigate";
178-
model?: "haiku";
185+
bounded?: true;
179186
}>("dashboard-agent.pendingActionTurn");
180187

181188
/**

‎internal-packages/dashboard-agent/src/compaction.test.ts‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,7 @@
11
// `@trigger.dev/sdk/ai/test` MUST be imported before the agent module so the
22
// resource catalog is installed before `chat.agent({ id })` registers.
33
import { mockChatAgent, type MockChatAgentHarness } from "@trigger.dev/sdk/ai/test";
4+
import { DASHBOARD_AGENT_SUMMARY_PROMPT } from "./prompts";
45

56
import { afterEach, describe, expect, it } from "vitest";
67
import { simulateReadableStream, type ModelMessage, type UIMessage } from "ai";
@@ -26,7 +27,6 @@ import {
2627
safeTail,
2728
shouldCompactConversation,
2829
STATIC_PREFIX_TOKENS,
29-
SUMMARY_INSTRUCTION,
3030
withDurableState,
3131
} from "./compaction";
3232

@@ -422,7 +422,7 @@ describe("the summariser's input", () => {
422422
* the watch line asks for a record and never for present state.
423423
*/
424424
describe("the summary instruction never asks for present state", () => {
425-
const watchLine = SUMMARY_INSTRUCTION.split("\n").find((line) => /watch/i.test(line));
425+
const watchLine = DASHBOARD_AGENT_SUMMARY_PROMPT.split("\n").find((line) => /watch/i.test(line));
426426

427427
it("has a line about watches at all", () => {
428428
expect(watchLine).toBeDefined();

‎internal-packages/dashboard-agent/src/compaction.ts‎

Lines changed: 22 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -9,6 +9,8 @@ import {
99
sanitizeReplayedToolInputs,
1010
} from "./agent-runtime";
1111
import { stripAgentLinks } from "./linkify-agent-text";
12+
import { dashboardAgentSummaryModel, promptModel, withoutThinking } from "./model-provider";
13+
import { summaryPrompt } from "./prompts";
1214

1315
/**
1416
* Bounded context: how a long conversation is summarised, and what may never be
@@ -54,26 +56,13 @@ export const COMPACTION_KEPT_TAIL_CHARS = 40_000;
5456
/** Per message, when the transcript is rendered for the summariser. */
5557
const SUMMARY_INPUT_MESSAGE_CHARS = 2_000;
5658

57-
/** The summariser: cheap, bounded, and told exactly what it may not drop. */
58-
const SUMMARY_MODEL = "anthropic:claude-haiku-4-5" as const;
59-
6059
/**
6160
* A hard ceiling on the summary, because "under 400 words" is an instruction and not a
6261
* budget. 400 words is ~530 tokens, so this is roughly double what the summary needs.
62+
* Thinking is switched off for the call, so none of it goes on hidden reasoning.
6363
*/
6464
const SUMMARY_MAX_OUTPUT_TOKENS = 1_000;
6565

66-
export const SUMMARY_INSTRUCTION = `You are compacting a support conversation between a user and an agent that reads a Trigger.dev dashboard, so the agent can keep going with a shorter history.
67-
68-
Write a summary in under 400 words, as notes rather than prose. Keep, in this order:
69-
1. What the user is trying to do, in their own terms, and anything they asked to be remembered.
70-
2. Facts already established, with the run ids, queue names, task identifiers, error fingerprints and numbers they rest on. Never restate a number you cannot see.
71-
3. Any investigation that is open: its investigationId, its title and its current outcome.
72-
4. Any watch the transcript records — what it was set up to watch, and what it said if it reported. Write it as what the transcript recorded, never as what is true now: a watch can expire or be cancelled without saying so here, so never present one as current.
73-
5. What was asked most recently and what is still unanswered.
74-
75-
Drop tool mechanics, retries, and anything already superseded. Do not add advice, and do not invent anything that is not in the transcript. Everything you write is a record of what the transcript said, not a claim about the present.`;
76-
7766
/** A summary that reads as a summary, and never as the user's next question. */
7867
function summaryMessage(summary: string, durableState?: string): ModelMessage {
7968
return {
@@ -273,11 +262,28 @@ export function renderTranscriptForSummary(messages: ModelMessage[]): string {
273262
}
274263

275264
async function summarizeConversation(event: SummarizeEvent): Promise<string> {
265+
// The summariser is a managed prompt: its text and model are versioned on the
266+
// platform and overridable from the dashboard, like the system prompt.
267+
const resolved = await summaryPrompt.resolve({});
268+
// Dashboard-managed call settings first; the bounded-call safeguards below stay fixed.
269+
const managed = (resolved.config ?? {}) as Partial<
270+
Pick<Parameters<typeof generateText>[0], "temperature" | "topP" | "topK" | "stopSequences">
271+
>;
276272
const { text } = await generateText({
277-
model: locals.get(dashboardAgentModelKey) ?? resolveDashboardAgentModel(SUMMARY_MODEL),
278-
system: SUMMARY_INSTRUCTION,
273+
...managed,
274+
model:
275+
locals.get(dashboardAgentModelKey) ??
276+
resolveDashboardAgentModel(
277+
promptModel(resolved, {
278+
env: "DASHBOARD_AGENT_SUMMARY_MODEL",
279+
fallback: dashboardAgentSummaryModel,
280+
})
281+
),
282+
system: resolved.text,
279283
prompt: renderTranscriptForSummary(event.messages),
280284
maxOutputTokens: SUMMARY_MAX_OUTPUT_TOKENS,
285+
providerOptions: withoutThinking(),
286+
...resolved.toAISDKTelemetry(),
281287
});
282288
return text.trim();
283289
}

‎internal-packages/dashboard-agent/src/dashboard-agent.eval.ts‎

Lines changed: 3 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -16,12 +16,13 @@ import {
1616
dashboardAgentToolsKey,
1717
type DashboardAgentStore,
1818
} from "./dashboard-agent";
19+
import { dashboardAgentJudgeModel, dashboardAgentModel } from "./model-provider";
1920
import { dashboardAgentCodeToolSchemas, dashboardAgentToolSchemas } from "./tool-schemas";
2021
import { showCodeAskPrompt } from "./tools";
2122

2223
const HAS_KEY = Boolean(process.env.ANTHROPIC_API_KEY);
23-
const AGENT_MODEL = "claude-sonnet-4-6";
24-
const JUDGE_MODEL = "claude-sonnet-4-6";
24+
const AGENT_MODEL = dashboardAgentModel();
25+
const JUDGE_MODEL = dashboardAgentJudgeModel();
2526

2627
const CLIENT_DATA = {
2728
userId: "user_eval",

‎internal-packages/dashboard-agent/src/dashboard-agent.ts‎

Lines changed: 38 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -20,6 +20,7 @@ import {
2020
dashboardAgentStorage,
2121
getStore,
2222
pendingActionTurnKey,
23+
turnModelKey,
2324
getSystemPrompt,
2425
modeFor,
2526
resolveDashboardAgentModel,
@@ -30,7 +31,14 @@ import {
3031
type DashboardAgentStore,
3132
} from "./agent-runtime";
3233
import { titlePrompt } from "./prompts";
33-
import { withCacheBreakpoint } from "./model-provider";
34+
import {
35+
dashboardAgentModel,
36+
dashboardAgentTitleModel,
37+
maxOutputTokensFor,
38+
promptModel,
39+
withCacheBreakpoint,
40+
withoutThinking,
41+
} from "./model-provider";
3442
import { recordPromptCacheUsage, stepCachePrepareStep } from "./step-cache";
3543
import { dashboardAgentActionSchema, handleWatchAction } from "./watch-actions";
3644
import { dashboardAgentCompaction, withDurableState } from "./compaction";
@@ -320,7 +328,12 @@ async function generateAndSaveTitle(
320328
const { text } = await generateText({
321329
model:
322330
locals.get(dashboardAgentModelKey) ??
323-
resolveDashboardAgentModel(resolved.model ?? "anthropic:claude-haiku-4-5"),
331+
resolveDashboardAgentModel(
332+
promptModel(resolved, {
333+
env: "DASHBOARD_AGENT_TITLE_MODEL",
334+
fallback: dashboardAgentTitleModel,
335+
})
336+
),
324337
system: resolved.text,
325338
prompt: userText,
326339
...resolved.toAISDKTelemetry(),
@@ -359,8 +372,8 @@ export function prepareTurnMessages(args: {
359372
);
360373
}
361374

362-
/** The output budget of a small-model wake: the headline and a sentence around it. */
363-
const HAIKU_WAKE_MAX_OUTPUT_TOKENS = 300;
375+
/** The output budget of an attention wake: the headline and a sentence around it. */
376+
const WAKE_MAX_OUTPUT_TOKENS = 300;
364377

365378
export const dashboardAgent = chat.agent({
366379
id: "dashboard-agent",
@@ -556,7 +569,9 @@ export const dashboardAgent = chat.agent({
556569
projectRef: clientData.projectRef,
557570
environment: clientData.environmentName,
558571
currentPage: clientData.currentPage,
559-
model: resolved.model,
572+
// The model that answered: an env override or dashboard override can
573+
// differ from the one the deployed prompt version names.
574+
model: locals.get(turnModelKey) ?? resolved.model,
560575
promptSlug: resolved.promptId,
561576
promptVersion: resolved.version,
562577
userText: userMessage ? extractText(userMessage) : "",
@@ -597,21 +612,28 @@ export const dashboardAgent = chat.agent({
597612
const actionTurn = locals.get(pendingActionTurnKey);
598613
const wake = actionTurn?.kind === "wake";
599614
const tools = wake ? {} : turnTools;
600-
// A wake the plan gave a headline is a sentence or two on the small model, bounded
601-
// so a chatty turn cannot run up the bill on something nobody asked.
602-
const smallWake = wake && actionTurn.model === "haiku";
615+
// A wake the plan gave a headline is a sentence or two, bounded so a chatty turn
616+
// cannot run up the bill on something nobody asked. Thinking is switched off for
617+
// it, so the cap is all answer.
618+
const boundedWake = wake && actionTurn.bounded === true;
603619
const options = chat.toStreamTextOptions({ tools });
620+
const modelId = promptModel(resolved, {
621+
env: "DASHBOARD_AGENT_MODEL",
622+
fallback: dashboardAgentModel,
623+
});
624+
locals.set(turnModelKey, modelId);
604625
let step = 0;
605626
return streamText({
606627
...options,
607-
model:
608-
locals.get(dashboardAgentModelKey) ??
609-
resolveDashboardAgentModel(
610-
smallWake
611-
? "anthropic:claude-haiku-4-5"
612-
: (resolved.model ?? "anthropic:claude-sonnet-4-6")
613-
),
614-
...(smallWake ? { maxOutputTokens: HAIKU_WAKE_MAX_OUTPUT_TOKENS } : {}),
628+
model: locals.get(dashboardAgentModelKey) ?? resolveDashboardAgentModel(modelId),
629+
// The documented ceiling for the model this turn resolved to; the pinned providers
630+
// would otherwise cap an unknown id at 4096, thinking included.
631+
...(maxOutputTokensFor(modelId) !== undefined
632+
? { maxOutputTokens: maxOutputTokensFor(modelId) }
633+
: {}),
634+
...(boundedWake
635+
? { maxOutputTokens: WAKE_MAX_OUTPUT_TOKENS, providerOptions: withoutThinking() }
636+
: {}),
615637
messages,
616638
abortSignal: signal,
617639
prepareStep: stepCachePrepareStep(options) as never,

‎internal-packages/dashboard-agent/src/eval-turn.ts‎

Lines changed: 4 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -5,7 +5,7 @@ import {
55
} from "@internal/dashboard-agent-db";
66
import { logger, task } from "@trigger.dev/sdk";
77
import { EVAL_ERROR_CATEGORIES, redactedEvalOutputErrored } from "./eval-policy";
8-
import { resolveDashboardAgentModel } from "./model-provider";
8+
import { resolveDashboardAgentModel, dashboardAgentJudgeModel } from "./model-provider";
99
import { generateObject } from "ai";
1010
import { z } from "zod";
1111

@@ -23,8 +23,6 @@ import { z } from "zod";
2323
* tool data verbatim. The comment above `evalTurn` lists exactly what a row holds.
2424
*/
2525

26-
const JUDGE_MODEL = "claude-sonnet-4-6";
27-
2826
// One connection pool per worker process for the eval task (separate from the
2927
// agent's; eval runs are their own runs and may land on other workers).
3028
let dbClient: DashboardAgentDbClient | undefined;
@@ -163,8 +161,9 @@ const JUDGE_SYSTEM = [
163161
export const evalTurn = task({
164162
id: "dashboard-agent-eval-turn",
165163
run: async (payload: EvalTurnPayload, { ctx }) => {
164+
const judgeModel = dashboardAgentJudgeModel();
166165
const { object } = await generateObject({
167-
model: resolveDashboardAgentModel(`anthropic:${JUDGE_MODEL}`),
166+
model: resolveDashboardAgentModel(`anthropic:${judgeModel}`),
168167
schema: TurnEval,
169168
system: JUDGE_SYSTEM,
170169
prompt: [
@@ -194,7 +193,7 @@ export const evalTurn = task({
194193
promptVersion: payload.promptVersion,
195194
toolsUsed: payload.toolActivity.map((t) => t.toolName),
196195
toolError,
197-
judgeModel: JUDGE_MODEL,
196+
judgeModel,
198197
scoreGrounded: object.grounded,
199198
scoreAnswered: object.answered,
200199
scoreConcise: object.concise,

0 commit comments

Comments
 (0)