Skip to content

Commit 04c8375

Browse files
ericallamTrigger.dev RepoOps
authored andcommitted
fix(sdk,dashboard-agent): stop a single tool-using turn compacting the chat conversation
A `chat.agent` with compaction configured could summarise a conversation after its very first question. The between-turns check was handed the turn's token usage summed over every tool-calling step, so a five-step turn reported its context five times over. It now receives the last step's usage, which is the context the model held on its final call. The summed figure is still on the event as `turnUsage`. A head-start handover that completes its pending tool call under the same message id now replaces the spliced partial in the model lane directly. Before, the replacement never matched and the lane was rebuilt from the transcript with a warning, which also dropped any pending lane injections. The dashboard agent decides compaction on the whole context the provider billed for the last call, 100k tokens by default and configurable with `DASHBOARD_AGENT_CONTEXT_TOKEN_BUDGET`. There is no longer a prefix constant to keep in step with the model or the prompt. Mono-RevId: 1ec83a82ec108975607b9aea4becd61deff1b6ee
1 parent dcb240e commit 04c8375

6 files changed

Lines changed: 290 additions & 65 deletions

File tree

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
1+
---
2+
"@trigger.dev/sdk": patch
3+
---
4+
5+
chat.agent: the between-turns compaction check now receives the last step's token usage (the context the model actually held) instead of the turn's sum over every tool-calling step, so a single tool-using turn no longer compacts a short conversation. The summed figure is still available as `turnUsage` on the event. A head-start handover whose pending tool call completes under the same message id now replaces its spliced partial in the model lane directly instead of falling back to a full reconversion.

‎internal-packages/dashboard-agent/src/compaction.test.ts‎

Lines changed: 34 additions & 13 deletions
Original file line numberDiff line numberDiff line change
@@ -20,13 +20,13 @@ import {
2020
collectDurableState,
2121
COMPACTION_KEPT_TAIL,
2222
COMPACTION_KEPT_TAIL_CHARS,
23-
CONVERSATION_TOKEN_BUDGET,
2423
describeDurableState,
2524
estimateConversationTokens,
2625
renderTranscriptForSummary,
2726
safeTail,
2827
shouldCompactConversation,
29-
STATIC_PREFIX_TOKENS,
28+
contextTokenBudget,
29+
DEFAULT_CONTEXT_TOKEN_BUDGET,
3030
withDurableState,
3131
} from "./compaction";
3232

@@ -146,24 +146,45 @@ describe("when the conversation is compacted", () => {
146146

147147
it("compacts on our own estimate, with no usage reported at all", () => {
148148
// 4 chars ≈ 1 token, so this is comfortably past the budget.
149-
const messages = bulk(40, (CONVERSATION_TOKEN_BUDGET * 4) / 20);
150-
expect(estimateConversationTokens(messages)).toBeGreaterThan(CONVERSATION_TOKEN_BUDGET);
149+
const messages = bulk(40, 12_000);
150+
expect(estimateConversationTokens(messages)).toBeGreaterThan(DEFAULT_CONTEXT_TOKEN_BUDGET);
151151
expect(shouldCompactConversation({ messages })).toBe(true);
152152
});
153153

154-
it("compacts on the provider's input count, net of the static prefix", () => {
154+
it("compacts on the context the provider billed for the last call", () => {
155+
const messages = bulk(4, 100);
156+
const budget = DEFAULT_CONTEXT_TOKEN_BUDGET;
157+
expect(shouldCompactConversation({ messages, inputTokens: 39_500 }, budget)).toBe(false);
158+
expect(shouldCompactConversation({ messages, inputTokens: budget }, budget)).toBe(false);
159+
expect(shouldCompactConversation({ messages, inputTokens: budget + 1 }, budget)).toBe(true);
160+
});
161+
162+
it("takes the context budget from the environment, falling back to the default", () => {
163+
expect(contextTokenBudget(undefined)).toBe(DEFAULT_CONTEXT_TOKEN_BUDGET);
164+
expect(contextTokenBudget(" 150000 ")).toBe(150_000);
165+
expect(contextTokenBudget("0")).toBe(DEFAULT_CONTEXT_TOKEN_BUDGET);
166+
expect(contextTokenBudget("lots")).toBe(DEFAULT_CONTEXT_TOKEN_BUDGET);
167+
expect(contextTokenBudget("1e5")).toBe(100_000);
168+
expect(contextTokenBudget("100k")).toBe(DEFAULT_CONTEXT_TOKEN_BUDGET);
169+
expect(contextTokenBudget("1.5")).toBe(DEFAULT_CONTEXT_TOKEN_BUDGET);
155170
const messages = bulk(4, 100);
156-
// The prefix alone must never trigger it.
157-
expect(shouldCompactConversation({ messages, inputTokens: STATIC_PREFIX_TOKENS + 100 })).toBe(
158-
false
159-
);
160171
expect(
161-
shouldCompactConversation({
162-
messages,
163-
inputTokens: STATIC_PREFIX_TOKENS + CONVERSATION_TOKEN_BUDGET + 1,
164-
})
172+
shouldCompactConversation({ messages, inputTokens: 50_001 }, contextTokenBudget("50000"))
165173
).toBe(true);
166174
});
175+
176+
it("lets a raised budget stand when the provider count is below it", () => {
177+
const messages = bulk(40, 12_000);
178+
expect(estimateConversationTokens(messages)).toBeGreaterThan(DEFAULT_CONTEXT_TOKEN_BUDGET);
179+
expect(estimateConversationTokens(messages)).toBeLessThan(150_000);
180+
expect(shouldCompactConversation({ messages, inputTokens: 120_000 }, 150_000)).toBe(false);
181+
expect(shouldCompactConversation({ messages, inputTokens: 150_001 }, 150_000)).toBe(true);
182+
});
183+
184+
it("does not compact a short conversation on a tool-using turn's summed usage", () => {
185+
const messages = bulk(12, 600);
186+
expect(shouldCompactConversation({ messages, inputTokens: 39_557 })).toBe(false);
187+
});
167188
});
168189

169190
describe("the state a summary may not swallow", () => {

‎internal-packages/dashboard-agent/src/compaction.ts‎

Lines changed: 35 additions & 23 deletions
Original file line numberDiff line numberDiff line change
@@ -28,20 +28,26 @@ import { summaryPrompt } from "./prompts";
2828
*/
2929

3030
/**
31-
* The static prefix (system prompt + tool schemas) as `prompt-prefix.test.ts`
32-
* measures it: ~20.9k estimated tokens. Subtracted so the budget below is about the
33-
* conversation rather than the total.
34-
*/
35-
export const STATIC_PREFIX_TOKENS = 21_000;
36-
37-
/**
38-
* How much conversation rides on top of that prefix before we summarise.
31+
* How large the model's context may grow before we summarise, as the provider
32+
* billed the LAST call of the turn: prefix, conversation and tool results together.
3933
*
40-
* 60k keeps a turn's input near 80k — well inside the 200k window even when a
41-
* 10-step turn's run traces and query rows add tens of thousands more — and it is
42-
* about where the uncached tail costs more per turn than one summary call does.
34+
* Budgeting the whole context rather than "input minus a prefix constant" means no
35+
* figure here has to track the model's tokenizer or the prompt's size (Sonnet 5
36+
* bills the same prefix ~1.8x what a chars/4 estimate gives). 100k is roughly the
37+
* old 60k of conversation on top of the measured ~38k prefix: well inside every
38+
* model's window even when a 10-step turn's run traces and query rows add tens of
39+
* thousands more, and about where the uncached tail costs more per turn than one
40+
* summary call does. `DASHBOARD_AGENT_CONTEXT_TOKEN_BUDGET` overrides it per
41+
* deployment.
4342
*/
44-
export const CONVERSATION_TOKEN_BUDGET = 60_000;
43+
export const DEFAULT_CONTEXT_TOKEN_BUDGET = 100_000;
44+
45+
export function contextTokenBudget(
46+
envValue = process.env.DASHBOARD_AGENT_CONTEXT_TOKEN_BUDGET
47+
): number {
48+
const parsed = Number(envValue?.trim() ?? "");
49+
return Number.isSafeInteger(parsed) && parsed > 0 ? parsed : DEFAULT_CONTEXT_TOKEN_BUDGET;
50+
}
4551

4652
/** Messages kept verbatim after the summary, so the last exchange reads normally. */
4753
export const COMPACTION_KEPT_TAIL = 8;
@@ -87,19 +93,24 @@ export function estimateConversationTokens(messages: ModelMessage[]): number {
8793
}
8894

8995
/**
90-
* Two signals, either of which fires: what the provider billed for input minus the
91-
* prefix, and our own estimate of the conversation. The estimate is what makes this
92-
* testable and what covers a call the provider reported no usage for.
96+
* Two signals, either of which fires against the same budget: the context the
97+
* provider billed on the last call, and our own chars/4 estimate of the conversation.
98+
* The estimate leaves out the prefix (it is not in `messages`) and so can only fire
99+
* later than the provider would; it covers a call with no reported usage and keeps
100+
* the decision testable without a provider. `inputTokens` must be the last step's,
101+
* not the turn's sum over steps, which `chat.agent` guarantees for both checks.
93102
*/
94-
export function shouldCompactConversation(event: {
95-
messages: ModelMessage[];
96-
inputTokens?: number;
97-
totalTokens?: number;
98-
}): boolean {
103+
export function shouldCompactConversation(
104+
event: {
105+
messages: ModelMessage[];
106+
inputTokens?: number;
107+
totalTokens?: number;
108+
},
109+
budget = contextTokenBudget()
110+
): boolean {
99111
const reported = typeof event.inputTokens === "number" ? event.inputTokens : event.totalTokens;
100-
const fromProvider = typeof reported === "number" ? reported - STATIC_PREFIX_TOKENS : 0;
101-
const estimated = estimateConversationTokens(event.messages);
102-
return Math.max(fromProvider, estimated) > CONVERSATION_TOKEN_BUDGET;
112+
if (typeof reported === "number" && reported > budget) return true;
113+
return estimateConversationTokens(event.messages) > budget;
103114
}
104115

105116
/* ------------------------------------------------------------------ *
@@ -304,6 +315,7 @@ export const dashboardAgentCompaction: ChatAgentCompactionOptions = {
304315
messageCount: event.messages.length,
305316
estimatedConversationTokens: estimateConversationTokens(event.messages),
306317
inputTokens: event.inputTokens ?? null,
318+
contextTokenBudget: contextTokenBudget(),
307319
});
308320
}
309321
return compact;

0 commit comments

Comments
 (0)