@vellumai/assistant 0.11.1-dev.202608041836.2a522b8 → 0.11.1-dev.202608042001.feb8dd6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/__tests__/compactor-truncation-pairing.test.ts +368 -0
- package/src/__tests__/history-repair.test.ts +39 -0
- package/src/agent/history-repair/history-repair.ts +8 -0
- package/src/context/compactor.ts +41 -4
- package/src/persistence/migrations/361-normalize-managed-connection-rows.test.ts +168 -0
- package/src/persistence/migrations/361-normalize-managed-connection-rows.ts +76 -0
- package/src/persistence/steps.ts +9 -0
- package/src/providers/__tests__/vellum-connection-routing.test.ts +5 -10
- package/src/providers/inference/adapter-factory.ts +4 -3
- package/src/providers/openai/__tests__/orphan-tool-result-guard.test.ts +478 -0
- package/src/providers/openai/chat-completions-provider.ts +46 -30
- package/src/providers/openai/orphaned-tool-result.ts +79 -0
- package/src/providers/openai/responses-provider.ts +50 -24
- package/src/providers/vellum-model-routing.ts +6 -7
- package/src/runtime/routes/inference-provider-connection-routes.ts +0 -16
package/package.json
CHANGED
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The compaction summary call's front truncation must never split a
|
|
3
|
+
* tool_use/tool_result pair, and the outbound request must be repaired
|
|
4
|
+
* before the provider call.
|
|
5
|
+
*
|
|
6
|
+
* The token-budget drop loop advances one message at a time, so without a
|
|
7
|
+
* boundary check its cut can drop an assistant `tool_use` while keeping its
|
|
8
|
+
* user `tool_result` in the retained portion. Providers that validate
|
|
9
|
+
* pairing (OpenAI Responses: "No tool call found for function call output
|
|
10
|
+
* with call_id ...") reject such a request, and the retry ladder reproduces
|
|
11
|
+
* the same rejection every pass. The cut must advance to a pair-safe user
|
|
12
|
+
* boundary, and `buildCompactionRequest` runs the deterministic history
|
|
13
|
+
* repair so even a malformed input history reaches the provider valid.
|
|
14
|
+
*/
|
|
15
|
+
import { describe, expect, mock, test } from "bun:test";
|
|
16
|
+
|
|
17
|
+
mock.module("../persistence/conversation-crud.js", () => ({
|
|
18
|
+
setConversationProcessingStartedAt: () => {},
|
|
19
|
+
isConversationProcessing: () => false,
|
|
20
|
+
getMessages: () => [],
|
|
21
|
+
}));
|
|
22
|
+
|
|
23
|
+
mock.module("../persistence/attachments-store.js", () => ({
|
|
24
|
+
getAttachmentMetadataForMessage: () => [],
|
|
25
|
+
getAttachmentContent: () => null,
|
|
26
|
+
}));
|
|
27
|
+
|
|
28
|
+
mock.module("../persistence/llm-request-log-store.js", () => ({
|
|
29
|
+
recordRequestLog: () => {},
|
|
30
|
+
}));
|
|
31
|
+
|
|
32
|
+
import { runAssistantDrivenCompaction } from "../context/compactor.js";
|
|
33
|
+
import { estimatePromptTokens } from "../context/token-estimator.js";
|
|
34
|
+
import type { ContentBlock, Message, Provider } from "../providers/types.js";
|
|
35
|
+
|
|
36
|
+
const SUMMARY = "Earlier turns summarized in the assistant's own voice.";
|
|
37
|
+
|
|
38
|
+
function turnTimestamp(turn: number): string {
|
|
39
|
+
const hour = String(10 + Math.floor(turn / 60)).padStart(2, "0");
|
|
40
|
+
const minute = String(turn % 60).padStart(2, "0");
|
|
41
|
+
return `2026-05-21 (Thursday) ${hour}:${minute}:00 -05:00 (America/Chicago)`;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function userTurn(turn: number, body: string): Message {
|
|
45
|
+
return {
|
|
46
|
+
role: "user",
|
|
47
|
+
content: [
|
|
48
|
+
{
|
|
49
|
+
type: "text",
|
|
50
|
+
text: `<turn_context>\ncurrent_time: ${turnTimestamp(
|
|
51
|
+
turn,
|
|
52
|
+
)}\n</turn_context>\n[U${turn}] ${body}`,
|
|
53
|
+
},
|
|
54
|
+
],
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function compactionResponse(tailTurn: number, preview: string): string {
|
|
59
|
+
return `<compaction_result>
|
|
60
|
+
<summary>
|
|
61
|
+
${SUMMARY}
|
|
62
|
+
</summary>
|
|
63
|
+
<key_state>
|
|
64
|
+
- Nothing critical pending.
|
|
65
|
+
</key_state>
|
|
66
|
+
<tail_start timestamp="${turnTimestamp(tailTurn)}" preview="${preview}" />
|
|
67
|
+
</compaction_result>`;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Fake provider enforcing the same pairing contract the OpenAI Responses
|
|
72
|
+
* serialization is subject to: a `tool_result` may only reference a
|
|
73
|
+
* `tool_use` emitted earlier in the same request (a `function_call_output`
|
|
74
|
+
* with no preceding matching `function_call` is a 400). Rejects with the
|
|
75
|
+
* provider's live error phrasing so a regression reproduces the real
|
|
76
|
+
* failure mode.
|
|
77
|
+
*/
|
|
78
|
+
function makeValidatingProvider(response: string): {
|
|
79
|
+
provider: Provider;
|
|
80
|
+
lastRequest: () => Message[] | null;
|
|
81
|
+
} {
|
|
82
|
+
let captured: Message[] | null = null;
|
|
83
|
+
const provider: Provider = {
|
|
84
|
+
name: "mock-provider",
|
|
85
|
+
sendMessage: async (messages: Message[]) => {
|
|
86
|
+
const emittedToolUseIds = new Set<string>();
|
|
87
|
+
for (const msg of messages) {
|
|
88
|
+
for (const block of msg.content) {
|
|
89
|
+
if (msg.role === "assistant" && block.type === "tool_use") {
|
|
90
|
+
emittedToolUseIds.add(block.id);
|
|
91
|
+
}
|
|
92
|
+
if (block.type === "tool_result") {
|
|
93
|
+
// guard:allow-tool-result-only: the fake validates client-side pairing only
|
|
94
|
+
const toolUseId = (block as { tool_use_id: string }).tool_use_id;
|
|
95
|
+
if (!emittedToolUseIds.has(toolUseId)) {
|
|
96
|
+
throw new Error(
|
|
97
|
+
`No tool call found for function call output with call_id ${toolUseId}.`,
|
|
98
|
+
);
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
captured = messages;
|
|
104
|
+
return {
|
|
105
|
+
content: [{ type: "text", text: response }],
|
|
106
|
+
model: "mock-model",
|
|
107
|
+
usage: { inputTokens: 100, outputTokens: 50 },
|
|
108
|
+
stopReason: "end_turn",
|
|
109
|
+
};
|
|
110
|
+
},
|
|
111
|
+
};
|
|
112
|
+
return { provider, lastRequest: () => captured };
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
function estimate(messages: Message[]): number {
|
|
116
|
+
return estimatePromptTokens(messages, "system", {
|
|
117
|
+
providerName: "mock-provider",
|
|
118
|
+
});
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/** Mirror of the compactor's pair-safe cut predicate for fixture guards. */
|
|
122
|
+
function isCleanUserBoundary(message: Message | undefined): boolean {
|
|
123
|
+
return (
|
|
124
|
+
message != null &&
|
|
125
|
+
message.role === "user" &&
|
|
126
|
+
// guard:allow-tool-result-only: mirrors the compactor's boundary predicate
|
|
127
|
+
!message.content.some((block) => block.type === "tool_result")
|
|
128
|
+
);
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Replay the token-budget drop loop without the pairing advance, to locate
|
|
133
|
+
* where the raw budget cut lands for a fixture.
|
|
134
|
+
*/
|
|
135
|
+
function naiveDropCount(messages: Message[], budgetTokens: number): number {
|
|
136
|
+
let dropCount = 0;
|
|
137
|
+
let estimated = estimate(messages);
|
|
138
|
+
while (estimated > budgetTokens && dropCount < messages.length - 1) {
|
|
139
|
+
dropCount++;
|
|
140
|
+
estimated = estimate(messages.slice(dropCount));
|
|
141
|
+
}
|
|
142
|
+
return dropCount;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
function textOfMessage(message: Message | undefined): string {
|
|
146
|
+
return (message?.content ?? [])
|
|
147
|
+
.map((block) => ("text" in block ? (block.text as string) : ""))
|
|
148
|
+
.join("\n");
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
function requestText(messages: Message[]): string {
|
|
152
|
+
return messages.map(textOfMessage).join("\n");
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
describe("compaction summary call: pair-safe front truncation", () => {
|
|
156
|
+
// Rounds of [user text, assistant tool_use (heavy), user tool_result,
|
|
157
|
+
// assistant text]: the heavy tool_use messages dominate the estimate, so
|
|
158
|
+
// the raw budget loop settles right after dropping one of them, landing
|
|
159
|
+
// the cut on its orphaned tool_result (or the assistant text behind it),
|
|
160
|
+
// never on a clean user boundary.
|
|
161
|
+
function buildToolHeavyHistory(rounds: number, idPrefix: string): Message[] {
|
|
162
|
+
const messages: Message[] = [];
|
|
163
|
+
for (let i = 0; i < rounds; i++) {
|
|
164
|
+
messages.push(userTurn(i, "please inspect the next data batch"));
|
|
165
|
+
messages.push({
|
|
166
|
+
role: "assistant",
|
|
167
|
+
content: [
|
|
168
|
+
{
|
|
169
|
+
type: "tool_use",
|
|
170
|
+
id: `${idPrefix}_${i}`,
|
|
171
|
+
name: "inspect_batch",
|
|
172
|
+
input: { payload: `batch ${i} `.repeat(600) },
|
|
173
|
+
},
|
|
174
|
+
],
|
|
175
|
+
});
|
|
176
|
+
messages.push({
|
|
177
|
+
role: "user",
|
|
178
|
+
content: [
|
|
179
|
+
{
|
|
180
|
+
type: "tool_result",
|
|
181
|
+
tool_use_id: `${idPrefix}_${i}`,
|
|
182
|
+
content: `inspection ${i} complete. `.repeat(8),
|
|
183
|
+
},
|
|
184
|
+
],
|
|
185
|
+
});
|
|
186
|
+
messages.push({
|
|
187
|
+
role: "assistant",
|
|
188
|
+
content: [
|
|
189
|
+
{ type: "text", text: `[A${i}] batch ${i} looks consistent.` },
|
|
190
|
+
],
|
|
191
|
+
});
|
|
192
|
+
}
|
|
193
|
+
return messages;
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
/** tool_result blocks in `messages` with no tool_use anywhere in `messages`. */
|
|
197
|
+
function orphanedResultIds(messages: Message[]): string[] {
|
|
198
|
+
const toolUseIds = new Set<string>();
|
|
199
|
+
for (const msg of messages) {
|
|
200
|
+
for (const block of msg.content) {
|
|
201
|
+
if (msg.role === "assistant" && block.type === "tool_use") {
|
|
202
|
+
toolUseIds.add(block.id);
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
const orphans: string[] = [];
|
|
207
|
+
for (const msg of messages) {
|
|
208
|
+
for (const block of msg.content) {
|
|
209
|
+
if (block.type === "tool_result") {
|
|
210
|
+
// guard:allow-tool-result-only: counting client-side orphans
|
|
211
|
+
const toolUseId = (block as { tool_use_id: string }).tool_use_id;
|
|
212
|
+
if (!toolUseIds.has(toolUseId)) {
|
|
213
|
+
orphans.push(toolUseId);
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
return orphans;
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
// Pairing must be id-format-agnostic, so the fixture runs across both
|
|
222
|
+
// persisted tool id shapes: Anthropic-style `toolu_` ids and OpenAI
|
|
223
|
+
// Responses-native `call_` ids.
|
|
224
|
+
for (const idPrefix of ["toolu", "call"] as const) {
|
|
225
|
+
test(`advances the cut to a pair-safe boundary and succeeds against a pairing-validating provider (${idPrefix}_ ids)`, async () => {
|
|
226
|
+
const rounds = 30;
|
|
227
|
+
const messages = buildToolHeavyHistory(rounds, idPrefix);
|
|
228
|
+
|
|
229
|
+
const maxInputTokens = 20_000;
|
|
230
|
+
// Mirrors compactor.compactionPrefixBudget: window minus the
|
|
231
|
+
// instruction reserve (800) and the 15% output reserve.
|
|
232
|
+
const prefixBudget =
|
|
233
|
+
maxInputTokens - 800 - Math.floor(maxInputTokens * 0.15);
|
|
234
|
+
expect(estimate(messages)).toBeGreaterThan(prefixBudget);
|
|
235
|
+
|
|
236
|
+
// Fixture guard: without the pairing advance, the budget cut lands on
|
|
237
|
+
// an unsafe index and the retained tail carries a tool_result whose
|
|
238
|
+
// tool_use sits in the dropped prefix (the exact shape the provider
|
|
239
|
+
// rejects). If token-estimation changes ever make this land safely,
|
|
240
|
+
// the fixture must be re-tuned or this test proves nothing.
|
|
241
|
+
const rawCut = naiveDropCount(messages, prefixBudget);
|
|
242
|
+
expect(rawCut).toBeGreaterThan(0);
|
|
243
|
+
expect(isCleanUserBoundary(messages[rawCut])).toBe(false);
|
|
244
|
+
expect(orphanedResultIds(messages.slice(rawCut)).length).toBeGreaterThan(
|
|
245
|
+
0,
|
|
246
|
+
);
|
|
247
|
+
|
|
248
|
+
const { provider, lastRequest } = makeValidatingProvider(
|
|
249
|
+
compactionResponse(rounds - 2, "please inspect the next"),
|
|
250
|
+
);
|
|
251
|
+
|
|
252
|
+
const result = await runAssistantDrivenCompaction({
|
|
253
|
+
conversationId: "conv-test",
|
|
254
|
+
messages,
|
|
255
|
+
provider,
|
|
256
|
+
systemPrompt: "system",
|
|
257
|
+
compaction: { enabled: true, autoThreshold: 0.7 },
|
|
258
|
+
maxInputTokens,
|
|
259
|
+
force: true,
|
|
260
|
+
previousEstimatedInputTokens: 90_000,
|
|
261
|
+
});
|
|
262
|
+
|
|
263
|
+
// The pairing-validating provider accepted the request and the pass
|
|
264
|
+
// applied.
|
|
265
|
+
expect(result.compacted).toBe(true);
|
|
266
|
+
|
|
267
|
+
const sent = lastRequest();
|
|
268
|
+
expect(sent).not.toBeNull();
|
|
269
|
+
const sentMessages = sent ?? [];
|
|
270
|
+
|
|
271
|
+
// Every tool_result in the outbound request is preceded by its
|
|
272
|
+
// tool_use.
|
|
273
|
+
expect(orphanedResultIds(sentMessages)).toEqual([]);
|
|
274
|
+
|
|
275
|
+
// The cut itself was pair-safe: nothing needed the repair pass's
|
|
276
|
+
// orphan downgrade, and the retained pairs survive intact.
|
|
277
|
+
const sentText = requestText(sentMessages);
|
|
278
|
+
expect(sentText).not.toContain("[orphaned");
|
|
279
|
+
|
|
280
|
+
// The request still fits the window and announces the truncation.
|
|
281
|
+
expect(estimate(sentMessages)).toBeLessThan(maxInputTokens);
|
|
282
|
+
expect(sentText).toContain("summary covers only the visible portion");
|
|
283
|
+
|
|
284
|
+
// Recent valid content is retained verbatim: the last round's user
|
|
285
|
+
// turn, tool pair, and assistant reply all reach the provider.
|
|
286
|
+
const lastRound = rounds - 1;
|
|
287
|
+
expect(sentText).toContain(`[U${lastRound}] please inspect`);
|
|
288
|
+
expect(sentText).toContain(`[A${lastRound}] batch ${lastRound}`);
|
|
289
|
+
const sentToolUseIds = new Set<string>();
|
|
290
|
+
for (const msg of sentMessages) {
|
|
291
|
+
for (const block of msg.content) {
|
|
292
|
+
if (msg.role === "assistant" && block.type === "tool_use") {
|
|
293
|
+
sentToolUseIds.add(block.id);
|
|
294
|
+
}
|
|
295
|
+
}
|
|
296
|
+
}
|
|
297
|
+
expect(sentToolUseIds.has(`${idPrefix}_${lastRound}`)).toBe(true);
|
|
298
|
+
});
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
test("repairs a malformed below-budget history before the provider call", async () => {
|
|
302
|
+
// Orphan tool_result (its tool_use exists nowhere in the history) and a
|
|
303
|
+
// consecutive same-role user run, below the truncation budget so the
|
|
304
|
+
// request-build repair is the only transform in play.
|
|
305
|
+
const messages: Message[] = [
|
|
306
|
+
userTurn(0, "start the job"),
|
|
307
|
+
{
|
|
308
|
+
role: "user",
|
|
309
|
+
content: [
|
|
310
|
+
{
|
|
311
|
+
type: "tool_result",
|
|
312
|
+
tool_use_id: "toolu_missing",
|
|
313
|
+
content: "stale result payload",
|
|
314
|
+
},
|
|
315
|
+
],
|
|
316
|
+
},
|
|
317
|
+
{
|
|
318
|
+
role: "assistant",
|
|
319
|
+
content: [{ type: "text", text: "[A0] acknowledged." }],
|
|
320
|
+
},
|
|
321
|
+
userTurn(1, "carry on with the job"),
|
|
322
|
+
{
|
|
323
|
+
role: "assistant",
|
|
324
|
+
content: [{ type: "text", text: "[A1] done." }],
|
|
325
|
+
},
|
|
326
|
+
];
|
|
327
|
+
|
|
328
|
+
const { provider, lastRequest } = makeValidatingProvider(
|
|
329
|
+
compactionResponse(1, "carry on with the job"),
|
|
330
|
+
);
|
|
331
|
+
|
|
332
|
+
const result = await runAssistantDrivenCompaction({
|
|
333
|
+
conversationId: "conv-test",
|
|
334
|
+
messages,
|
|
335
|
+
provider,
|
|
336
|
+
systemPrompt: "system",
|
|
337
|
+
compaction: { enabled: true, autoThreshold: 0.7 },
|
|
338
|
+
maxInputTokens: 200_000,
|
|
339
|
+
force: true,
|
|
340
|
+
previousEstimatedInputTokens: 90_000,
|
|
341
|
+
});
|
|
342
|
+
|
|
343
|
+
expect(result.compacted).toBe(true);
|
|
344
|
+
|
|
345
|
+
const sentMessages = lastRequest() ?? [];
|
|
346
|
+
// The orphan was downgraded to text: no tool_result blocks reach the
|
|
347
|
+
// provider, and the payload is preserved in the degraded form.
|
|
348
|
+
const hasToolResult = sentMessages.some((msg) =>
|
|
349
|
+
// guard:allow-tool-result-only: asserting the repair removed them
|
|
350
|
+
msg.content.some((block: ContentBlock) => block.type === "tool_result"),
|
|
351
|
+
);
|
|
352
|
+
expect(hasToolResult).toBe(false);
|
|
353
|
+
const sentText = requestText(sentMessages);
|
|
354
|
+
expect(sentText).toContain("stale result payload");
|
|
355
|
+
expect(sentText).toContain("orphaned tool_result");
|
|
356
|
+
|
|
357
|
+
// Same-role runs were merged: roles strictly alternate in the history
|
|
358
|
+
// portion of the request (everything before the trailing instruction).
|
|
359
|
+
const historyPortion = sentMessages.slice(0, -1);
|
|
360
|
+
for (let i = 1; i < historyPortion.length; i++) {
|
|
361
|
+
expect(historyPortion[i].role).not.toBe(historyPortion[i - 1].role);
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
// Valid user text survives the repair verbatim.
|
|
365
|
+
expect(sentText).toContain("[U0] start the job");
|
|
366
|
+
expect(sentText).toContain("[U1] carry on with the job");
|
|
367
|
+
});
|
|
368
|
+
});
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { describe, expect, test } from "bun:test";
|
|
2
2
|
|
|
3
3
|
import { deepRepairHistory } from "../agent/history-repair/history-repair.js";
|
|
4
|
+
import { isRepairableOrderingError } from "../agent/history-repair/history-repair.js";
|
|
4
5
|
import { repairHistory } from "../agent/history-repair/history-repair.js";
|
|
5
6
|
import type { Message } from "../providers/types.js";
|
|
6
7
|
|
|
@@ -1110,3 +1111,41 @@ describe("deepRepairHistory", () => {
|
|
|
1110
1111
|
expect(stats.orphanToolResultsDowngraded).toBe(0);
|
|
1111
1112
|
});
|
|
1112
1113
|
});
|
|
1114
|
+
|
|
1115
|
+
describe("isRepairableOrderingError", () => {
|
|
1116
|
+
test("matches Anthropic tool ordering rejections", () => {
|
|
1117
|
+
expect(
|
|
1118
|
+
isRepairableOrderingError(
|
|
1119
|
+
"Requests which include `tool_use` blocks must have a corresponding `tool_result` block in the next message.",
|
|
1120
|
+
),
|
|
1121
|
+
).toBe(true);
|
|
1122
|
+
expect(
|
|
1123
|
+
isRepairableOrderingError(
|
|
1124
|
+
"messages.5: tool_result block references tool_use_id toolu_abc which was not found",
|
|
1125
|
+
),
|
|
1126
|
+
).toBe(true);
|
|
1127
|
+
});
|
|
1128
|
+
|
|
1129
|
+
test("matches the OpenAI Responses orphan function_call_output rejection", () => {
|
|
1130
|
+
expect(
|
|
1131
|
+
isRepairableOrderingError(
|
|
1132
|
+
"No tool call found for function call output with call_id call_abc123.",
|
|
1133
|
+
),
|
|
1134
|
+
).toBe(true);
|
|
1135
|
+
});
|
|
1136
|
+
|
|
1137
|
+
test("matches the OpenAI Chat Completions orphan tool_call_id rejection", () => {
|
|
1138
|
+
expect(
|
|
1139
|
+
isRepairableOrderingError(
|
|
1140
|
+
"Invalid parameter: 'tool_call_id' of 'call_abc123' not found in 'tool_calls' of previous message.",
|
|
1141
|
+
),
|
|
1142
|
+
).toBe(true);
|
|
1143
|
+
});
|
|
1144
|
+
|
|
1145
|
+
test("does not match unrelated provider errors", () => {
|
|
1146
|
+
expect(isRepairableOrderingError("Rate limit exceeded")).toBe(false);
|
|
1147
|
+
expect(
|
|
1148
|
+
isRepairableOrderingError("Request too large for model context window"),
|
|
1149
|
+
).toBe(false);
|
|
1150
|
+
});
|
|
1151
|
+
});
|
|
@@ -399,6 +399,14 @@ export const ORDERING_ERROR_PATTERNS: readonly RegExp[] = [
|
|
|
399
399
|
/tool_use_id.*without.*tool_result/i,
|
|
400
400
|
/tool_result.*tool_use_id.*not found/i,
|
|
401
401
|
/messages.*invalid.*order/i,
|
|
402
|
+
// OpenAI Responses API: a function_call_output whose call_id has no
|
|
403
|
+
// matching function_call earlier in the request ("No tool call found for
|
|
404
|
+
// function call output with call_id ...").
|
|
405
|
+
/no tool call found for function call output/i,
|
|
406
|
+
// OpenAI Chat Completions API: a tool message whose tool_call_id is not in
|
|
407
|
+
// a preceding assistant message's tool_calls ("Invalid parameter:
|
|
408
|
+
// 'tool_call_id' of '...' not found in 'tool_calls' of previous message").
|
|
409
|
+
/tool_call_id.*not found/i,
|
|
402
410
|
];
|
|
403
411
|
|
|
404
412
|
/**
|
package/src/context/compactor.ts
CHANGED
|
@@ -18,6 +18,7 @@
|
|
|
18
18
|
* On any parse or resolution failure we abort the compaction and return
|
|
19
19
|
* `compacted: false` — never silently lose messages.
|
|
20
20
|
*/
|
|
21
|
+
import { repairHistory } from "../agent/history-repair/history-repair.js";
|
|
21
22
|
import { optimizeImageForTransport } from "../agent/image-optimize.js";
|
|
22
23
|
import type { CompactionConfig } from "../config/schemas/compaction.js";
|
|
23
24
|
import type { LLMCallSite } from "../config/schemas/llm.js";
|
|
@@ -1056,9 +1057,14 @@ function extractTextFromResponse(content: ContentBlock[]): string {
|
|
|
1056
1057
|
|
|
1057
1058
|
// Build the outbound message list for a compaction provider call: apply the
|
|
1058
1059
|
// same pre-send sanitization bundle as the agent loop's model calls
|
|
1059
|
-
// (`preModelCallSanitize
|
|
1060
|
-
// collapsed, historical web-search results converted to text),
|
|
1061
|
-
//
|
|
1060
|
+
// (`preModelCallSanitize`: old tool-result media stripped, AX trees
|
|
1061
|
+
// collapsed, historical web-search results converted to text), run the
|
|
1062
|
+
// deterministic history repair over the sanitized projection, then append
|
|
1063
|
+
// the summarization instruction at the tail. The repair pass downgrades any
|
|
1064
|
+
// orphaned `tool_result` (its `tool_use` outside the request, e.g. cut off
|
|
1065
|
+
// by front truncation) to plain text and merges consecutive same-role runs,
|
|
1066
|
+
// so the request always satisfies the provider's pairing validation. A
|
|
1067
|
+
// well-formed history passes through repair structurally unchanged.
|
|
1062
1068
|
//
|
|
1063
1069
|
// Matching the loop's projection matters for two reasons. First, the summary
|
|
1064
1070
|
// call's prefix stays byte-aligned with the agent's warm prompt cache — an
|
|
@@ -1076,7 +1082,10 @@ function buildCompactionRequest(
|
|
|
1076
1082
|
history: Message[],
|
|
1077
1083
|
instruction: Message,
|
|
1078
1084
|
): Message[] {
|
|
1079
|
-
return [
|
|
1085
|
+
return [
|
|
1086
|
+
...repairHistory(preModelCallSanitize(history)).messages,
|
|
1087
|
+
instruction,
|
|
1088
|
+
];
|
|
1080
1089
|
}
|
|
1081
1090
|
|
|
1082
1091
|
// Token headroom a compaction summary call reserves on top of its history: room
|
|
@@ -1129,6 +1138,34 @@ function truncateHistoryToBudget(args: {
|
|
|
1129
1138
|
if (dropCount === 0) {
|
|
1130
1139
|
return messages;
|
|
1131
1140
|
}
|
|
1141
|
+
// Advance the cut to a pair-safe boundary. The budget loop stops wherever
|
|
1142
|
+
// the estimate first fits, which can land between an assistant `tool_use`
|
|
1143
|
+
// and its user `tool_result` and leave an orphaned `tool_result` opening
|
|
1144
|
+
// the retained portion (rejected by providers that validate pairing).
|
|
1145
|
+
// Walk forward to the next clean user boundary, never dropping the final
|
|
1146
|
+
// message (mirroring the budget loop's own bound). When no boundary exists
|
|
1147
|
+
// the requested cut stands; the request-build repair pass downgrades any
|
|
1148
|
+
// orphaned results so the outbound call remains valid.
|
|
1149
|
+
const requestedDropCount = dropCount;
|
|
1150
|
+
if (!isForwardCutBoundary(messages, dropCount)) {
|
|
1151
|
+
for (let i = dropCount + 1; i < messages.length; i++) {
|
|
1152
|
+
if (isForwardCutBoundary(messages, i)) {
|
|
1153
|
+
dropCount = i;
|
|
1154
|
+
break;
|
|
1155
|
+
}
|
|
1156
|
+
}
|
|
1157
|
+
}
|
|
1158
|
+
if (dropCount !== requestedDropCount) {
|
|
1159
|
+
log.info(
|
|
1160
|
+
{
|
|
1161
|
+
requestedDropCount,
|
|
1162
|
+
pairSafeDropCount: dropCount,
|
|
1163
|
+
budgetTokens,
|
|
1164
|
+
totalMessages: messages.length,
|
|
1165
|
+
},
|
|
1166
|
+
"Advanced compaction summary-call front truncation to a pair-safe boundary",
|
|
1167
|
+
);
|
|
1168
|
+
}
|
|
1132
1169
|
log.info(
|
|
1133
1170
|
{ dropCount, budgetTokens, totalMessages: messages.length },
|
|
1134
1171
|
"Compaction summary input exceeds context window — truncating from front",
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
import { Database } from "bun:sqlite";
|
|
2
|
+
import { describe, expect, test } from "bun:test";
|
|
3
|
+
|
|
4
|
+
import { drizzle } from "drizzle-orm/bun-sqlite";
|
|
5
|
+
|
|
6
|
+
import * as schema from "../schema.js";
|
|
7
|
+
import { migrateNormalizeManagedConnectionRows } from "./361-normalize-managed-connection-rows.js";
|
|
8
|
+
|
|
9
|
+
function createTestDb() {
|
|
10
|
+
const sqlite = new Database(":memory:");
|
|
11
|
+
sqlite.exec(/*sql*/ `
|
|
12
|
+
CREATE TABLE provider_connections (
|
|
13
|
+
name TEXT PRIMARY KEY,
|
|
14
|
+
provider TEXT NOT NULL,
|
|
15
|
+
auth TEXT NOT NULL,
|
|
16
|
+
label TEXT,
|
|
17
|
+
base_url TEXT,
|
|
18
|
+
models TEXT,
|
|
19
|
+
created_at INTEGER NOT NULL,
|
|
20
|
+
updated_at INTEGER NOT NULL
|
|
21
|
+
);
|
|
22
|
+
`);
|
|
23
|
+
return { sqlite, db: drizzle(sqlite, { schema }) };
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
function insertRow(
|
|
27
|
+
sqlite: Database,
|
|
28
|
+
name: string,
|
|
29
|
+
provider: string,
|
|
30
|
+
auth: string,
|
|
31
|
+
): void {
|
|
32
|
+
sqlite
|
|
33
|
+
.query(
|
|
34
|
+
/*sql*/ `INSERT INTO provider_connections (name, provider, auth, created_at, updated_at)
|
|
35
|
+
VALUES (?, ?, ?, 1, 1)`,
|
|
36
|
+
)
|
|
37
|
+
.run(name, provider, auth);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function readRows(
|
|
41
|
+
sqlite: Database,
|
|
42
|
+
): Array<{ name: string; provider: string; auth: string }> {
|
|
43
|
+
return sqlite
|
|
44
|
+
.query(
|
|
45
|
+
`SELECT name, provider, auth FROM provider_connections ORDER BY name`,
|
|
46
|
+
)
|
|
47
|
+
.all() as Array<{ name: string; provider: string; auth: string }>;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
function readRow(
|
|
51
|
+
sqlite: Database,
|
|
52
|
+
name: string,
|
|
53
|
+
): { provider: string; auth: string } {
|
|
54
|
+
const row = readRows(sqlite).find((r) => r.name === name);
|
|
55
|
+
if (!row) {
|
|
56
|
+
throw new Error(`row "${name}" not found`);
|
|
57
|
+
}
|
|
58
|
+
return row;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
describe("migration 361: normalize managed connection rows", () => {
|
|
62
|
+
test("rewrites a platform-auth row with a concrete provider to provider vellum", () => {
|
|
63
|
+
const { sqlite, db } = createTestDb();
|
|
64
|
+
insertRow(sqlite, "managed-openai", "openai", '{"type":"platform"}');
|
|
65
|
+
|
|
66
|
+
migrateNormalizeManagedConnectionRows(db);
|
|
67
|
+
|
|
68
|
+
const row = readRow(sqlite, "managed-openai");
|
|
69
|
+
expect(row.provider).toBe("vellum");
|
|
70
|
+
expect(JSON.parse(row.auth)).toEqual({ type: "platform" });
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
test("heals a stuck legacy canonical row (name vellum, concrete provider, platform auth)", () => {
|
|
74
|
+
const { sqlite, db } = createTestDb();
|
|
75
|
+
insertRow(sqlite, "vellum", "anthropic", '{"type":"platform"}');
|
|
76
|
+
|
|
77
|
+
migrateNormalizeManagedConnectionRows(db);
|
|
78
|
+
|
|
79
|
+
expect(readRow(sqlite, "vellum").provider).toBe("vellum");
|
|
80
|
+
});
|
|
81
|
+
|
|
82
|
+
test("rewrites a vellum-provider row with non-platform auth to platform auth", () => {
|
|
83
|
+
const { sqlite, db } = createTestDb();
|
|
84
|
+
insertRow(
|
|
85
|
+
sqlite,
|
|
86
|
+
"keyed-vellum",
|
|
87
|
+
"vellum",
|
|
88
|
+
'{"type":"api_key","credential":"vault/x"}',
|
|
89
|
+
);
|
|
90
|
+
insertRow(sqlite, "none-vellum", "vellum", '{"type":"none"}');
|
|
91
|
+
|
|
92
|
+
migrateNormalizeManagedConnectionRows(db);
|
|
93
|
+
|
|
94
|
+
expect(JSON.parse(readRow(sqlite, "keyed-vellum").auth)).toEqual({
|
|
95
|
+
type: "platform",
|
|
96
|
+
});
|
|
97
|
+
expect(JSON.parse(readRow(sqlite, "none-vellum").auth)).toEqual({
|
|
98
|
+
type: "platform",
|
|
99
|
+
});
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
test("leaves consistent rows untouched", () => {
|
|
103
|
+
const { sqlite, db } = createTestDb();
|
|
104
|
+
insertRow(sqlite, "vellum", "vellum", '{"type":"platform"}');
|
|
105
|
+
insertRow(
|
|
106
|
+
sqlite,
|
|
107
|
+
"anthropic-key",
|
|
108
|
+
"anthropic",
|
|
109
|
+
'{"type":"api_key","credential":"vault/anthropic"}',
|
|
110
|
+
);
|
|
111
|
+
insertRow(
|
|
112
|
+
sqlite,
|
|
113
|
+
"chatgpt-subscription",
|
|
114
|
+
"openai",
|
|
115
|
+
'{"type":"oauth_subscription"}',
|
|
116
|
+
);
|
|
117
|
+
insertRow(sqlite, "ollama-local", "ollama", '{"type":"none"}');
|
|
118
|
+
|
|
119
|
+
const before = readRows(sqlite);
|
|
120
|
+
migrateNormalizeManagedConnectionRows(db);
|
|
121
|
+
|
|
122
|
+
expect(readRows(sqlite)).toEqual(before);
|
|
123
|
+
});
|
|
124
|
+
|
|
125
|
+
test("leaves a BYOK row claiming the canonical name untouched", () => {
|
|
126
|
+
const { sqlite, db } = createTestDb();
|
|
127
|
+
insertRow(
|
|
128
|
+
sqlite,
|
|
129
|
+
"vellum",
|
|
130
|
+
"anthropic",
|
|
131
|
+
'{"type":"api_key","credential":"vault/anthropic"}',
|
|
132
|
+
);
|
|
133
|
+
|
|
134
|
+
migrateNormalizeManagedConnectionRows(db);
|
|
135
|
+
|
|
136
|
+
const row = readRow(sqlite, "vellum");
|
|
137
|
+
expect(row.provider).toBe("anthropic");
|
|
138
|
+
expect(JSON.parse(row.auth).type).toBe("api_key");
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
test("leaves a row with unparseable auth untouched", () => {
|
|
142
|
+
const { sqlite, db } = createTestDb();
|
|
143
|
+
insertRow(sqlite, "broken", "vellum", "not json");
|
|
144
|
+
|
|
145
|
+
migrateNormalizeManagedConnectionRows(db);
|
|
146
|
+
|
|
147
|
+
expect(readRow(sqlite, "broken").auth).toBe("not json");
|
|
148
|
+
});
|
|
149
|
+
|
|
150
|
+
test("is idempotent", () => {
|
|
151
|
+
const { sqlite, db } = createTestDb();
|
|
152
|
+
insertRow(sqlite, "managed-openai", "openai", '{"type":"platform"}');
|
|
153
|
+
insertRow(sqlite, "keyed-vellum", "vellum", '{"type":"api_key"}');
|
|
154
|
+
|
|
155
|
+
migrateNormalizeManagedConnectionRows(db);
|
|
156
|
+
const afterFirst = readRows(sqlite);
|
|
157
|
+
migrateNormalizeManagedConnectionRows(db);
|
|
158
|
+
|
|
159
|
+
expect(readRows(sqlite)).toEqual(afterFirst);
|
|
160
|
+
});
|
|
161
|
+
|
|
162
|
+
test("no-ops on a database from before the provider_connections table", () => {
|
|
163
|
+
const sqlite = new Database(":memory:");
|
|
164
|
+
const db = drizzle(sqlite, { schema });
|
|
165
|
+
|
|
166
|
+
expect(() => migrateNormalizeManagedConnectionRows(db)).not.toThrow();
|
|
167
|
+
});
|
|
168
|
+
});
|